| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.022071502471548453, |
| "eval_steps": 500, |
| "global_step": 384, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15703.0, |
| "completions/max_terminated_length": 15703.0, |
| "completions/mean_length": 8732.875, |
| "completions/mean_terminated_length": 8732.875, |
| "completions/min_length": 4987.0, |
| "completions/min_terminated_length": 4987.0, |
| "entropy": 0.3713345378637314, |
| "epoch": 5.7477871019657434e-05, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 70583.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000192403793335, |
| "sampling/importance_sampling_ratio/min": 1.836119345455245e-08, |
| "sampling/sampling_logp_difference/max": 17.813026428222656, |
| "sampling/sampling_logp_difference/mean": 0.015137896873056889, |
| "step": 1 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15439.0, |
| "completions/mean_length": 12614.125, |
| "completions/mean_terminated_length": 10352.2001953125, |
| "completions/min_length": 6948.0, |
| "completions/min_terminated_length": 6948.0, |
| "entropy": 0.36503184773027897, |
| "epoch": 0.00011495574203931487, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012826107442378998, |
| "learning_rate": 1e-05, |
| "loss": -0.0246, |
| "num_tokens": 172536.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999308586120605, |
| "sampling/importance_sampling_ratio/min": 0.12457425892353058, |
| "sampling/sampling_logp_difference/max": 2.082853317260742, |
| "sampling/sampling_logp_difference/mean": 0.016638126224279404, |
| "step": 2 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2265.0, |
| "completions/max_terminated_length": 2265.0, |
| "completions/mean_length": 1789.5, |
| "completions/mean_terminated_length": 1789.5, |
| "completions/min_length": 1057.0, |
| "completions/min_terminated_length": 1057.0, |
| "entropy": 0.2469680141657591, |
| "epoch": 0.0001724336130589723, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 187676.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4334642887115479, |
| "sampling/importance_sampling_ratio/mean": 0.9997435808181763, |
| "sampling/importance_sampling_ratio/min": 0.7285618185997009, |
| "sampling/sampling_logp_difference/max": 0.3600940704345703, |
| "sampling/sampling_logp_difference/mean": 0.00893494300544262, |
| "step": 3 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7182.0, |
| "completions/max_terminated_length": 7182.0, |
| "completions/mean_length": 2169.125, |
| "completions/mean_terminated_length": 2169.125, |
| "completions/min_length": 722.0, |
| "completions/min_terminated_length": 722.0, |
| "entropy": 0.20379706658422947, |
| "epoch": 0.00022991148407862974, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 205845.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8169389963150024, |
| "sampling/importance_sampling_ratio/mean": 1.0000684261322021, |
| "sampling/importance_sampling_ratio/min": 0.6421703696250916, |
| "sampling/sampling_logp_difference/max": 0.5971531867980957, |
| "sampling/sampling_logp_difference/mean": 0.008074535056948662, |
| "step": 4 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6306.0, |
| "completions/max_terminated_length": 6306.0, |
| "completions/mean_length": 2645.0, |
| "completions/mean_terminated_length": 2645.0, |
| "completions/min_length": 1777.0, |
| "completions/min_terminated_length": 1777.0, |
| "entropy": 0.25945289619266987, |
| "epoch": 0.0002873893550982872, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 227981.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.912467122077942, |
| "sampling/importance_sampling_ratio/mean": 1.0001256465911865, |
| "sampling/importance_sampling_ratio/min": 0.6602612733840942, |
| "sampling/sampling_logp_difference/max": 0.6483941078186035, |
| "sampling/sampling_logp_difference/mean": 0.007415128871798515, |
| "step": 5 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014546304009854794, |
| "clip_ratio/high_mean": 0.00014546304009854794, |
| "clip_ratio/low_mean": 0.00022717010870110244, |
| "clip_ratio/low_min": 0.00022717010870110244, |
| "clip_ratio/region_mean": 0.00037263314879965037, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7348.0, |
| "completions/max_terminated_length": 7348.0, |
| "completions/mean_length": 5479.375, |
| "completions/mean_terminated_length": 5479.375, |
| "completions/min_length": 3462.0, |
| "completions/min_terminated_length": 3462.0, |
| "entropy": 0.2911783792078495, |
| "epoch": 0.0003448672261179446, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010940776206552982, |
| "learning_rate": 1e-05, |
| "loss": -0.1459, |
| "num_tokens": 272792.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.6280268430709839, |
| "sampling/importance_sampling_ratio/mean": 0.9998860359191895, |
| "sampling/importance_sampling_ratio/min": 0.574209988117218, |
| "sampling/sampling_logp_difference/max": 0.5547601580619812, |
| "sampling/sampling_logp_difference/mean": 0.011181830428540707, |
| "step": 6 |
| }, |
| { |
| "clip_ratio/high_max": 5.0924933020723984e-05, |
| "clip_ratio/high_mean": 5.0924933020723984e-05, |
| "clip_ratio/low_mean": 0.00038385049265343696, |
| "clip_ratio/low_min": 0.00038385049265343696, |
| "clip_ratio/region_mean": 0.00043477542567416094, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 11842.0, |
| "completions/mean_length": 9243.0, |
| "completions/mean_terminated_length": 8222.857421875, |
| "completions/min_length": 3075.0, |
| "completions/min_terminated_length": 3075.0, |
| "entropy": 0.4120573028922081, |
| "epoch": 0.000402345097137602, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009335801936686039, |
| "learning_rate": 1e-05, |
| "loss": 0.099, |
| "num_tokens": 348000.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.8537434339523315, |
| "sampling/importance_sampling_ratio/mean": 1.0000827312469482, |
| "sampling/importance_sampling_ratio/min": 0.3109295070171356, |
| "sampling/sampling_logp_difference/max": 1.1681890487670898, |
| "sampling/sampling_logp_difference/mean": 0.0167313814163208, |
| "step": 7 |
| }, |
| { |
| "clip_ratio/high_max": 0.00019611547213571612, |
| "clip_ratio/high_mean": 0.00019611547213571612, |
| "clip_ratio/low_mean": 0.00048802775563672185, |
| "clip_ratio/low_min": 0.00048802775563672185, |
| "clip_ratio/region_mean": 0.000684143227772438, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14617.0, |
| "completions/max_terminated_length": 14617.0, |
| "completions/mean_length": 9390.625, |
| "completions/mean_terminated_length": 9390.625, |
| "completions/min_length": 4170.0, |
| "completions/min_terminated_length": 4170.0, |
| "entropy": 0.45750724896788597, |
| "epoch": 0.00045982296815725947, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008300665766000748, |
| "learning_rate": 1e-05, |
| "loss": -0.0132, |
| "num_tokens": 424301.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000351667404175, |
| "sampling/importance_sampling_ratio/min": 0.3896910846233368, |
| "sampling/sampling_logp_difference/max": 0.9424009323120117, |
| "sampling/sampling_logp_difference/mean": 0.0190004613250494, |
| "step": 8 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15432.0, |
| "completions/mean_length": 13763.0, |
| "completions/mean_terminated_length": 11142.0, |
| "completions/min_length": 7535.0, |
| "completions/min_terminated_length": 7535.0, |
| "entropy": 0.5409758687019348, |
| "epoch": 0.0005173008391769169, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 535325.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997733235359192, |
| "sampling/importance_sampling_ratio/min": 0.2753318250179291, |
| "sampling/sampling_logp_difference/max": 1.289778232574463, |
| "sampling/sampling_logp_difference/mean": 0.02159600332379341, |
| "step": 9 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00023255814448930323, |
| "clip_ratio/low_min": 0.00023255814448930323, |
| "clip_ratio/region_mean": 0.00023255814448930323, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3956.0, |
| "completions/max_terminated_length": 3956.0, |
| "completions/mean_length": 2248.5, |
| "completions/mean_terminated_length": 2248.5, |
| "completions/min_length": 705.0, |
| "completions/min_terminated_length": 705.0, |
| "entropy": 0.5861556529998779, |
| "epoch": 0.0005747787101965744, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.029013216495513916, |
| "learning_rate": 1e-05, |
| "loss": -0.2064, |
| "num_tokens": 554473.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.374398946762085, |
| "sampling/importance_sampling_ratio/mean": 0.9997981190681458, |
| "sampling/importance_sampling_ratio/min": 0.6573039889335632, |
| "sampling/sampling_logp_difference/max": 0.4196087121963501, |
| "sampling/sampling_logp_difference/mean": 0.015369054861366749, |
| "step": 10 |
| }, |
| { |
| "clip_ratio/high_max": 6.59255929349456e-05, |
| "clip_ratio/high_mean": 6.59255929349456e-05, |
| "clip_ratio/low_mean": 0.0005529337686311919, |
| "clip_ratio/low_min": 0.0005529337686311919, |
| "clip_ratio/region_mean": 0.0006188593615661375, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14214.0, |
| "completions/mean_length": 13965.75, |
| "completions/mean_terminated_length": 11547.5, |
| "completions/min_length": 8701.0, |
| "completions/min_terminated_length": 8701.0, |
| "entropy": 0.6653081439435482, |
| "epoch": 0.0006322565812162317, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009686644189059734, |
| "learning_rate": 1e-05, |
| "loss": 0.0451, |
| "num_tokens": 667743.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999228119850159, |
| "sampling/importance_sampling_ratio/min": 0.0002580129657872021, |
| "sampling/sampling_logp_difference/max": 8.262500762939453, |
| "sampling/sampling_logp_difference/mean": 0.02505280077457428, |
| "step": 11 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12835.0, |
| "completions/mean_length": 10192.125, |
| "completions/mean_terminated_length": 9307.572265625, |
| "completions/min_length": 3177.0, |
| "completions/min_terminated_length": 3177.0, |
| "entropy": 0.5226390846073627, |
| "epoch": 0.0006897344522358892, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 751376.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999798536300659, |
| "sampling/importance_sampling_ratio/min": 0.11669129133224487, |
| "sampling/sampling_logp_difference/max": 2.148223400115967, |
| "sampling/sampling_logp_difference/mean": 0.020940717309713364, |
| "step": 12 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13361.0, |
| "completions/mean_length": 8044.25, |
| "completions/mean_terminated_length": 6852.857421875, |
| "completions/min_length": 2809.0, |
| "completions/min_terminated_length": 2809.0, |
| "entropy": 0.527798768132925, |
| "epoch": 0.0007472123232555466, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 817834.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999900460243225, |
| "sampling/importance_sampling_ratio/min": 0.31529974937438965, |
| "sampling/sampling_logp_difference/max": 1.1542315483093262, |
| "sampling/sampling_logp_difference/mean": 0.023021675646305084, |
| "step": 13 |
| }, |
| { |
| "clip_ratio/high_max": 8.7596352386754e-05, |
| "clip_ratio/high_mean": 8.7596352386754e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 8.7596352386754e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1512.0, |
| "completions/max_terminated_length": 1512.0, |
| "completions/mean_length": 984.375, |
| "completions/mean_terminated_length": 984.375, |
| "completions/min_length": 503.0, |
| "completions/min_terminated_length": 503.0, |
| "entropy": 0.36184784211218357, |
| "epoch": 0.000804690194275204, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024298641830682755, |
| "learning_rate": 1e-05, |
| "loss": 0.1052, |
| "num_tokens": 826485.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.2456766366958618, |
| "sampling/importance_sampling_ratio/mean": 0.9998593926429749, |
| "sampling/importance_sampling_ratio/min": 0.32329079508781433, |
| "sampling/sampling_logp_difference/max": 1.1292030811309814, |
| "sampling/sampling_logp_difference/mean": 0.012408224865794182, |
| "step": 14 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001685967235971475, |
| "clip_ratio/high_mean": 0.0001685967235971475, |
| "clip_ratio/low_mean": 0.00014894512423779815, |
| "clip_ratio/low_min": 0.00014894512423779815, |
| "clip_ratio/region_mean": 0.00031754184783494566, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14267.0, |
| "completions/max_terminated_length": 14267.0, |
| "completions/mean_length": 6477.0, |
| "completions/mean_terminated_length": 6477.0, |
| "completions/min_length": 2448.0, |
| "completions/min_terminated_length": 2448.0, |
| "entropy": 0.4685293845832348, |
| "epoch": 0.0008621680652948615, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014940561726689339, |
| "learning_rate": 1e-05, |
| "loss": 0.1567, |
| "num_tokens": 879269.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8391656875610352, |
| "sampling/importance_sampling_ratio/mean": 1.0000338554382324, |
| "sampling/importance_sampling_ratio/min": 0.4191240668296814, |
| "sampling/sampling_logp_difference/max": 0.8695883750915527, |
| "sampling/sampling_logp_difference/mean": 0.019313381984829903, |
| "step": 15 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4386.0, |
| "completions/max_terminated_length": 4386.0, |
| "completions/mean_length": 3328.875, |
| "completions/mean_terminated_length": 3328.875, |
| "completions/min_length": 2100.0, |
| "completions/min_terminated_length": 2100.0, |
| "entropy": 0.3531157746911049, |
| "epoch": 0.0009196459363145189, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 907412.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4276796579360962, |
| "sampling/importance_sampling_ratio/mean": 0.9996469616889954, |
| "sampling/importance_sampling_ratio/min": 0.6109430193901062, |
| "sampling/sampling_logp_difference/max": 0.4927515983581543, |
| "sampling/sampling_logp_difference/mean": 0.013146447017788887, |
| "step": 16 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0005994474158796947, |
| "clip_ratio/low_min": 0.0005994474158796947, |
| "clip_ratio/region_mean": 0.0005994474158796947, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13680.0, |
| "completions/mean_length": 10788.0, |
| "completions/mean_terminated_length": 9988.572265625, |
| "completions/min_length": 6727.0, |
| "completions/min_terminated_length": 6727.0, |
| "entropy": 0.3596680872142315, |
| "epoch": 0.0009771238073341764, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00487625552341342, |
| "learning_rate": 1e-05, |
| "loss": 0.1051, |
| "num_tokens": 995540.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.926710605621338, |
| "sampling/importance_sampling_ratio/mean": 1.0001296997070312, |
| "sampling/importance_sampling_ratio/min": 0.2658487856388092, |
| "sampling/sampling_logp_difference/max": 1.3248276710510254, |
| "sampling/sampling_logp_difference/mean": 0.01606706902384758, |
| "step": 17 |
| }, |
| { |
| "clip_ratio/high_max": 4.931870898872148e-05, |
| "clip_ratio/high_mean": 4.931870898872148e-05, |
| "clip_ratio/low_mean": 0.00017867315182229504, |
| "clip_ratio/low_min": 0.00017867315182229504, |
| "clip_ratio/region_mean": 0.00022799186081101652, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9202.0, |
| "completions/max_terminated_length": 9202.0, |
| "completions/mean_length": 4511.375, |
| "completions/mean_terminated_length": 4511.375, |
| "completions/min_length": 1294.0, |
| "completions/min_terminated_length": 1294.0, |
| "entropy": 0.3361959885805845, |
| "epoch": 0.0010346016783538338, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021912362426519394, |
| "learning_rate": 1e-05, |
| "loss": -0.2679, |
| "num_tokens": 1033319.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000478029251099, |
| "sampling/importance_sampling_ratio/min": 0.589799702167511, |
| "sampling/sampling_logp_difference/max": 0.7484335899353027, |
| "sampling/sampling_logp_difference/mean": 0.013660256750881672, |
| "step": 18 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7835.0, |
| "completions/max_terminated_length": 7835.0, |
| "completions/mean_length": 3494.5, |
| "completions/mean_terminated_length": 3494.5, |
| "completions/min_length": 860.0, |
| "completions/min_terminated_length": 860.0, |
| "entropy": 0.26188469864428043, |
| "epoch": 0.0010920795493734913, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 1062099.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000929832458496, |
| "sampling/importance_sampling_ratio/min": 0.5309983491897583, |
| "sampling/sampling_logp_difference/max": 1.0871529579162598, |
| "sampling/sampling_logp_difference/mean": 0.010413752868771553, |
| "step": 19 |
| }, |
| { |
| "clip_ratio/high_max": 1.1474206075945403e-05, |
| "clip_ratio/high_mean": 1.1474206075945403e-05, |
| "clip_ratio/low_mean": 0.000845233731524786, |
| "clip_ratio/low_min": 0.000845233731524786, |
| "clip_ratio/region_mean": 0.0008567079376007314, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15349.0, |
| "completions/mean_length": 12602.125, |
| "completions/mean_terminated_length": 12061.857421875, |
| "completions/min_length": 8332.0, |
| "completions/min_terminated_length": 8332.0, |
| "entropy": 0.7199657186865807, |
| "epoch": 0.0011495574203931487, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007488382514566183, |
| "learning_rate": 1e-05, |
| "loss": 0.0481, |
| "num_tokens": 1164124.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002334117889404, |
| "sampling/importance_sampling_ratio/min": 0.02026776410639286, |
| "sampling/sampling_logp_difference/max": 3.898723602294922, |
| "sampling/sampling_logp_difference/mean": 0.028150716796517372, |
| "step": 20 |
| }, |
| { |
| "clip_ratio/high_max": 3.3866159355966374e-05, |
| "clip_ratio/high_mean": 3.3866159355966374e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 3.3866159355966374e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3691.0, |
| "completions/max_terminated_length": 3691.0, |
| "completions/mean_length": 2487.75, |
| "completions/mean_terminated_length": 2487.75, |
| "completions/min_length": 1564.0, |
| "completions/min_terminated_length": 1564.0, |
| "entropy": 0.30037602595984936, |
| "epoch": 0.0012070352914128062, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011537784710526466, |
| "learning_rate": 1e-05, |
| "loss": -0.0959, |
| "num_tokens": 1185050.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.5541020631790161, |
| "sampling/importance_sampling_ratio/mean": 1.0000855922698975, |
| "sampling/importance_sampling_ratio/min": 0.7744051814079285, |
| "sampling/sampling_logp_difference/max": 0.44089794158935547, |
| "sampling/sampling_logp_difference/mean": 0.011413312517106533, |
| "step": 21 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0001804557723517064, |
| "clip_ratio/low_min": 0.0001804557723517064, |
| "clip_ratio/region_mean": 0.0001804557723517064, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11208.0, |
| "completions/max_terminated_length": 11208.0, |
| "completions/mean_length": 5979.5, |
| "completions/mean_terminated_length": 5979.5, |
| "completions/min_length": 2086.0, |
| "completions/min_terminated_length": 2086.0, |
| "entropy": 0.4145108927041292, |
| "epoch": 0.0012645131624324634, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01590256206691265, |
| "learning_rate": 1e-05, |
| "loss": 0.0982, |
| "num_tokens": 1233846.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8413699865341187, |
| "sampling/importance_sampling_ratio/mean": 0.9999732971191406, |
| "sampling/importance_sampling_ratio/min": 0.41654255986213684, |
| "sampling/sampling_logp_difference/max": 0.8757666349411011, |
| "sampling/sampling_logp_difference/mean": 0.013827912509441376, |
| "step": 22 |
| }, |
| { |
| "clip_ratio/high_max": 1.0090409887197893e-05, |
| "clip_ratio/high_mean": 1.0090409887197893e-05, |
| "clip_ratio/low_mean": 7.030371489236131e-05, |
| "clip_ratio/low_min": 7.030371489236131e-05, |
| "clip_ratio/region_mean": 8.03941247795592e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12388.0, |
| "completions/max_terminated_length": 12388.0, |
| "completions/mean_length": 4933.875, |
| "completions/mean_terminated_length": 4933.875, |
| "completions/min_length": 1821.0, |
| "completions/min_terminated_length": 1821.0, |
| "entropy": 0.27186551690101624, |
| "epoch": 0.0013219910334521209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010686766356229782, |
| "learning_rate": 1e-05, |
| "loss": -0.0986, |
| "num_tokens": 1274613.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.8778114318847656, |
| "sampling/importance_sampling_ratio/mean": 1.000512957572937, |
| "sampling/importance_sampling_ratio/min": 0.5995104312896729, |
| "sampling/sampling_logp_difference/max": 0.6301069259643555, |
| "sampling/sampling_logp_difference/mean": 0.011764852330088615, |
| "step": 23 |
| }, |
| { |
| "clip_ratio/high_max": 0.00021059082064311951, |
| "clip_ratio/high_mean": 0.00021059082064311951, |
| "clip_ratio/low_mean": 0.00010199918324360624, |
| "clip_ratio/low_min": 0.00010199918324360624, |
| "clip_ratio/region_mean": 0.00031259000388672575, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15383.0, |
| "completions/max_terminated_length": 15383.0, |
| "completions/mean_length": 9550.625, |
| "completions/mean_terminated_length": 9550.625, |
| "completions/min_length": 6705.0, |
| "completions/min_terminated_length": 6705.0, |
| "entropy": 0.4893578588962555, |
| "epoch": 0.0013794689044717783, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007482054177671671, |
| "learning_rate": 1e-05, |
| "loss": 0.0095, |
| "num_tokens": 1351946.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999520778656006, |
| "sampling/importance_sampling_ratio/min": 0.3111985921859741, |
| "sampling/sampling_logp_difference/max": 1.1673240661621094, |
| "sampling/sampling_logp_difference/mean": 0.019285906106233597, |
| "step": 24 |
| }, |
| { |
| "clip_ratio/high_max": 6.948304508114234e-05, |
| "clip_ratio/high_mean": 6.948304508114234e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.948304508114234e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1799.0, |
| "completions/max_terminated_length": 1799.0, |
| "completions/mean_length": 1389.0, |
| "completions/mean_terminated_length": 1389.0, |
| "completions/min_length": 855.0, |
| "completions/min_terminated_length": 855.0, |
| "entropy": 0.1900008488446474, |
| "epoch": 0.0014369467754914358, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010644371621310711, |
| "learning_rate": 1e-05, |
| "loss": 0.0707, |
| "num_tokens": 1364522.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.2926770448684692, |
| "sampling/importance_sampling_ratio/mean": 1.0001492500305176, |
| "sampling/importance_sampling_ratio/min": 0.7533293962478638, |
| "sampling/sampling_logp_difference/max": 0.2832527160644531, |
| "sampling/sampling_logp_difference/mean": 0.0072412146255373955, |
| "step": 25 |
| }, |
| { |
| "clip_ratio/high_max": 0.00011978824113612063, |
| "clip_ratio/high_mean": 0.00011978824113612063, |
| "clip_ratio/low_mean": 9.366803715238348e-05, |
| "clip_ratio/low_min": 9.366803715238348e-05, |
| "clip_ratio/region_mean": 0.0002134562782885041, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3192.0, |
| "completions/max_terminated_length": 3192.0, |
| "completions/mean_length": 2180.625, |
| "completions/mean_terminated_length": 2180.625, |
| "completions/min_length": 1240.0, |
| "completions/min_terminated_length": 1240.0, |
| "entropy": 0.37862563878297806, |
| "epoch": 0.0014944246465110932, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03671795129776001, |
| "learning_rate": 1e-05, |
| "loss": 0.0621, |
| "num_tokens": 1382967.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4113205671310425, |
| "sampling/importance_sampling_ratio/mean": 0.9998127818107605, |
| "sampling/importance_sampling_ratio/min": 0.49675899744033813, |
| "sampling/sampling_logp_difference/max": 0.6996502876281738, |
| "sampling/sampling_logp_difference/mean": 0.014092679135501385, |
| "step": 26 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6224.0, |
| "completions/max_terminated_length": 6224.0, |
| "completions/mean_length": 3049.125, |
| "completions/mean_terminated_length": 3049.125, |
| "completions/min_length": 1252.0, |
| "completions/min_terminated_length": 1252.0, |
| "entropy": 1.1191394180059433, |
| "epoch": 0.0015519025175307506, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 1409560.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6623228788375854, |
| "sampling/importance_sampling_ratio/mean": 0.9997996091842651, |
| "sampling/importance_sampling_ratio/min": 0.6972482204437256, |
| "sampling/sampling_logp_difference/max": 0.5082159042358398, |
| "sampling/sampling_logp_difference/mean": 0.023721016943454742, |
| "step": 27 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017029972514137626, |
| "clip_ratio/high_mean": 0.00017029972514137626, |
| "clip_ratio/low_mean": 0.0003643309319159016, |
| "clip_ratio/low_min": 0.0003643309319159016, |
| "clip_ratio/region_mean": 0.0005346306570572779, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11496.0, |
| "completions/max_terminated_length": 11496.0, |
| "completions/mean_length": 2998.0, |
| "completions/mean_terminated_length": 2998.0, |
| "completions/min_length": 734.0, |
| "completions/min_terminated_length": 734.0, |
| "entropy": 0.35346287302672863, |
| "epoch": 0.001609380388550408, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.026703374460339546, |
| "learning_rate": 1e-05, |
| "loss": 0.4945, |
| "num_tokens": 1436712.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.8510832786560059, |
| "sampling/importance_sampling_ratio/mean": 1.0000603199005127, |
| "sampling/importance_sampling_ratio/min": 0.4190898835659027, |
| "sampling/sampling_logp_difference/max": 0.8696699142456055, |
| "sampling/sampling_logp_difference/mean": 0.016062427312135696, |
| "step": 28 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7258.0, |
| "completions/max_terminated_length": 7258.0, |
| "completions/mean_length": 5022.75, |
| "completions/mean_terminated_length": 5022.75, |
| "completions/min_length": 2704.0, |
| "completions/min_terminated_length": 2704.0, |
| "entropy": 0.47850726172327995, |
| "epoch": 0.0016668582595700655, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 1478150.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5747973918914795, |
| "sampling/importance_sampling_ratio/mean": 0.9998484253883362, |
| "sampling/importance_sampling_ratio/min": 0.2432888150215149, |
| "sampling/sampling_logp_difference/max": 1.413506031036377, |
| "sampling/sampling_logp_difference/mean": 0.012552469037473202, |
| "step": 29 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0006041391061444301, |
| "clip_ratio/low_min": 0.0006041391061444301, |
| "clip_ratio/region_mean": 0.0006041391061444301, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4163.0, |
| "completions/max_terminated_length": 4163.0, |
| "completions/mean_length": 2496.25, |
| "completions/mean_terminated_length": 2496.25, |
| "completions/min_length": 1023.0, |
| "completions/min_terminated_length": 1023.0, |
| "entropy": 0.4747072793543339, |
| "epoch": 0.001724336130589723, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02458236925303936, |
| "learning_rate": 1e-05, |
| "loss": -0.0758, |
| "num_tokens": 1499672.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4631445407867432, |
| "sampling/importance_sampling_ratio/mean": 1.0004292726516724, |
| "sampling/importance_sampling_ratio/min": 0.695888876914978, |
| "sampling/sampling_logp_difference/max": 0.38058793544769287, |
| "sampling/sampling_logp_difference/mean": 0.01702776364982128, |
| "step": 30 |
| }, |
| { |
| "clip_ratio/high_max": 0.00011294151954643894, |
| "clip_ratio/high_mean": 0.00011294151954643894, |
| "clip_ratio/low_mean": 0.00013817244325764477, |
| "clip_ratio/low_min": 0.00013817244325764477, |
| "clip_ratio/region_mean": 0.0002511139628040837, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9644.0, |
| "completions/max_terminated_length": 9644.0, |
| "completions/mean_length": 4240.625, |
| "completions/mean_terminated_length": 4240.625, |
| "completions/min_length": 2661.0, |
| "completions/min_terminated_length": 2661.0, |
| "entropy": 0.40254703164100647, |
| "epoch": 0.0017818140016093804, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011770940385758877, |
| "learning_rate": 1e-05, |
| "loss": -0.1578, |
| "num_tokens": 1535061.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001137256622314, |
| "sampling/importance_sampling_ratio/min": 0.5766231417655945, |
| "sampling/sampling_logp_difference/max": 0.8971240520477295, |
| "sampling/sampling_logp_difference/mean": 0.015235140919685364, |
| "step": 31 |
| }, |
| { |
| "clip_ratio/high_max": 3.31272094626911e-05, |
| "clip_ratio/high_mean": 3.31272094626911e-05, |
| "clip_ratio/low_mean": 0.00022139876818982884, |
| "clip_ratio/low_min": 0.00022139876818982884, |
| "clip_ratio/region_mean": 0.00025452597765251994, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11392.0, |
| "completions/max_terminated_length": 11392.0, |
| "completions/mean_length": 4060.75, |
| "completions/mean_terminated_length": 4060.75, |
| "completions/min_length": 1180.0, |
| "completions/min_terminated_length": 1180.0, |
| "entropy": 0.37329025007784367, |
| "epoch": 0.0018392918726290379, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02815098874270916, |
| "learning_rate": 1e-05, |
| "loss": 0.1951, |
| "num_tokens": 1568635.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000230073928833, |
| "sampling/importance_sampling_ratio/min": 0.5119144916534424, |
| "sampling/sampling_logp_difference/max": 0.7037672996520996, |
| "sampling/sampling_logp_difference/mean": 0.019624225795269012, |
| "step": 32 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017782688155421056, |
| "clip_ratio/high_mean": 0.00017782688155421056, |
| "clip_ratio/low_mean": 0.0002970715577248484, |
| "clip_ratio/low_min": 0.0002970715577248484, |
| "clip_ratio/region_mean": 0.00047489843927905895, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13661.0, |
| "completions/max_terminated_length": 13661.0, |
| "completions/mean_length": 7598.5, |
| "completions/mean_terminated_length": 7598.5, |
| "completions/min_length": 4159.0, |
| "completions/min_terminated_length": 4159.0, |
| "entropy": 0.6333885863423347, |
| "epoch": 0.0018967697436486953, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013582466170191765, |
| "learning_rate": 1e-05, |
| "loss": 0.1817, |
| "num_tokens": 1630687.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998713731765747, |
| "sampling/importance_sampling_ratio/min": 0.30169931054115295, |
| "sampling/sampling_logp_difference/max": 1.19832444190979, |
| "sampling/sampling_logp_difference/mean": 0.02295018918812275, |
| "step": 33 |
| }, |
| { |
| "clip_ratio/high_max": 2.9680441002710722e-05, |
| "clip_ratio/high_mean": 2.9680441002710722e-05, |
| "clip_ratio/low_mean": 0.0005807479028590024, |
| "clip_ratio/low_min": 0.0005807479028590024, |
| "clip_ratio/region_mean": 0.0006104283438617131, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15896.0, |
| "completions/mean_length": 12444.75, |
| "completions/mean_terminated_length": 10081.2001953125, |
| "completions/min_length": 5550.0, |
| "completions/min_terminated_length": 5550.0, |
| "entropy": 0.4394209682941437, |
| "epoch": 0.0019542476146683528, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0108946543186903, |
| "learning_rate": 1e-05, |
| "loss": 0.1747, |
| "num_tokens": 1731557.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999930202960968, |
| "sampling/importance_sampling_ratio/min": 0.16880448162555695, |
| "sampling/sampling_logp_difference/max": 1.7790141105651855, |
| "sampling/sampling_logp_difference/mean": 0.01957680657505989, |
| "step": 34 |
| }, |
| { |
| "clip_ratio/high_max": 7.717753396718763e-05, |
| "clip_ratio/high_mean": 7.717753396718763e-05, |
| "clip_ratio/low_mean": 0.00036224109135218896, |
| "clip_ratio/low_min": 0.00036224109135218896, |
| "clip_ratio/region_mean": 0.0004394186253193766, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 10658.0, |
| "completions/mean_length": 7760.375, |
| "completions/mean_terminated_length": 6528.4287109375, |
| "completions/min_length": 3282.0, |
| "completions/min_terminated_length": 3282.0, |
| "entropy": 0.3291672505438328, |
| "epoch": 0.00201172548568801, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013807074166834354, |
| "learning_rate": 1e-05, |
| "loss": 0.2426, |
| "num_tokens": 1794696.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.7015570402145386, |
| "sampling/importance_sampling_ratio/mean": 0.9997872114181519, |
| "sampling/importance_sampling_ratio/min": 0.2956359386444092, |
| "sampling/sampling_logp_difference/max": 1.2186264991760254, |
| "sampling/sampling_logp_difference/mean": 0.015166893601417542, |
| "step": 35 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00026345426158513874, |
| "clip_ratio/low_min": 0.00026345426158513874, |
| "clip_ratio/region_mean": 0.00026345426158513874, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1544.0, |
| "completions/max_terminated_length": 1544.0, |
| "completions/mean_length": 1201.125, |
| "completions/mean_terminated_length": 1201.125, |
| "completions/min_length": 731.0, |
| "completions/min_terminated_length": 731.0, |
| "entropy": 0.273766353726387, |
| "epoch": 0.0020692033567076677, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03457275405526161, |
| "learning_rate": 1e-05, |
| "loss": -0.0332, |
| "num_tokens": 1805449.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.281903862953186, |
| "sampling/importance_sampling_ratio/mean": 0.9998401999473572, |
| "sampling/importance_sampling_ratio/min": 0.7044704556465149, |
| "sampling/sampling_logp_difference/max": 0.350308895111084, |
| "sampling/sampling_logp_difference/mean": 0.010058763436973095, |
| "step": 36 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0006667066845693626, |
| "clip_ratio/low_min": 0.0006667066845693626, |
| "clip_ratio/region_mean": 0.0006667066845693626, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 11499.0, |
| "completions/mean_length": 9040.0, |
| "completions/mean_terminated_length": 6592.0, |
| "completions/min_length": 2660.0, |
| "completions/min_terminated_length": 2660.0, |
| "entropy": 0.45479580760002136, |
| "epoch": 0.002126681227727325, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.012254394590854645, |
| "learning_rate": 1e-05, |
| "loss": 0.3725, |
| "num_tokens": 1879545.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999937415122986, |
| "sampling/importance_sampling_ratio/min": 0.1515624076128006, |
| "sampling/sampling_logp_difference/max": 1.8867578506469727, |
| "sampling/sampling_logp_difference/mean": 0.018890954554080963, |
| "step": 37 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012891687038063537, |
| "clip_ratio/high_mean": 0.00012891687038063537, |
| "clip_ratio/low_mean": 0.0004333783217589371, |
| "clip_ratio/low_min": 0.0004333783217589371, |
| "clip_ratio/region_mean": 0.0005622951921395725, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16173.0, |
| "completions/max_terminated_length": 16173.0, |
| "completions/mean_length": 9202.375, |
| "completions/mean_terminated_length": 9202.375, |
| "completions/min_length": 2678.0, |
| "completions/min_terminated_length": 2678.0, |
| "entropy": 0.38314999639987946, |
| "epoch": 0.0021841590987469826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01417507790029049, |
| "learning_rate": 1e-05, |
| "loss": 0.0678, |
| "num_tokens": 1954084.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000090599060059, |
| "sampling/importance_sampling_ratio/min": 0.2041807472705841, |
| "sampling/sampling_logp_difference/max": 1.588749647140503, |
| "sampling/sampling_logp_difference/mean": 0.01605384796857834, |
| "step": 38 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001823197799240006, |
| "clip_ratio/high_mean": 0.0001823197799240006, |
| "clip_ratio/low_mean": 0.0003852306108456105, |
| "clip_ratio/low_min": 0.0003852306108456105, |
| "clip_ratio/region_mean": 0.0005675503907696111, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14687.0, |
| "completions/max_terminated_length": 14687.0, |
| "completions/mean_length": 8147.0, |
| "completions/mean_terminated_length": 8147.0, |
| "completions/min_length": 916.0, |
| "completions/min_terminated_length": 916.0, |
| "entropy": 0.46798358857631683, |
| "epoch": 0.00224163696976664, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025270069018006325, |
| "learning_rate": 1e-05, |
| "loss": -0.3309, |
| "num_tokens": 2020420.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.9406111240386963, |
| "sampling/importance_sampling_ratio/mean": 1.0002000331878662, |
| "sampling/importance_sampling_ratio/min": 0.36602357029914856, |
| "sampling/sampling_logp_difference/max": 1.0050575733184814, |
| "sampling/sampling_logp_difference/mean": 0.020551156252622604, |
| "step": 39 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 9.865824540611356e-05, |
| "clip_ratio/low_min": 9.865824540611356e-05, |
| "clip_ratio/region_mean": 9.865824540611356e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1972.0, |
| "completions/max_terminated_length": 1972.0, |
| "completions/mean_length": 1393.875, |
| "completions/mean_terminated_length": 1393.875, |
| "completions/min_length": 918.0, |
| "completions/min_terminated_length": 918.0, |
| "entropy": 0.3229250833392143, |
| "epoch": 0.0022991148407862975, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025179797783493996, |
| "learning_rate": 1e-05, |
| "loss": 0.0512, |
| "num_tokens": 2032915.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4177850484848022, |
| "sampling/importance_sampling_ratio/mean": 1.0002374649047852, |
| "sampling/importance_sampling_ratio/min": 0.7231569290161133, |
| "sampling/sampling_logp_difference/max": 0.34909582138061523, |
| "sampling/sampling_logp_difference/mean": 0.01290525309741497, |
| "step": 40 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015636347234249115, |
| "clip_ratio/high_mean": 0.00015636347234249115, |
| "clip_ratio/low_mean": 0.00015422578144352883, |
| "clip_ratio/low_min": 0.00015422578144352883, |
| "clip_ratio/region_mean": 0.00031058925378602, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7163.0, |
| "completions/max_terminated_length": 7163.0, |
| "completions/mean_length": 5693.625, |
| "completions/mean_terminated_length": 5693.625, |
| "completions/min_length": 3660.0, |
| "completions/min_terminated_length": 3660.0, |
| "entropy": 0.6593504920601845, |
| "epoch": 0.0023565927118059547, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009168462827801704, |
| "learning_rate": 1e-05, |
| "loss": 0.0491, |
| "num_tokens": 2079400.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6397300958633423, |
| "sampling/importance_sampling_ratio/mean": 1.0000221729278564, |
| "sampling/importance_sampling_ratio/min": 0.5670493841171265, |
| "sampling/sampling_logp_difference/max": 0.5673089027404785, |
| "sampling/sampling_logp_difference/mean": 0.023446809500455856, |
| "step": 41 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3320.0, |
| "completions/max_terminated_length": 3320.0, |
| "completions/mean_length": 2165.625, |
| "completions/mean_terminated_length": 2165.625, |
| "completions/min_length": 1036.0, |
| "completions/min_terminated_length": 1036.0, |
| "entropy": 0.4291406534612179, |
| "epoch": 0.0024140705828256124, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2097741.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4754000902175903, |
| "sampling/importance_sampling_ratio/mean": 1.0001964569091797, |
| "sampling/importance_sampling_ratio/min": 0.6778994202613831, |
| "sampling/sampling_logp_difference/max": 0.38892924785614014, |
| "sampling/sampling_logp_difference/mean": 0.01713942177593708, |
| "step": 42 |
| }, |
| { |
| "clip_ratio/high_max": 6.959911115700379e-05, |
| "clip_ratio/high_mean": 6.959911115700379e-05, |
| "clip_ratio/low_mean": 0.0003096781365456991, |
| "clip_ratio/low_min": 0.0003096781365456991, |
| "clip_ratio/region_mean": 0.0003792772477027029, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2918.0, |
| "completions/max_terminated_length": 2918.0, |
| "completions/mean_length": 2122.125, |
| "completions/mean_terminated_length": 2122.125, |
| "completions/min_length": 1466.0, |
| "completions/min_terminated_length": 1466.0, |
| "entropy": 0.2825094908475876, |
| "epoch": 0.0024715484538452696, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.029945719987154007, |
| "learning_rate": 1e-05, |
| "loss": 0.1292, |
| "num_tokens": 2115766.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4234094619750977, |
| "sampling/importance_sampling_ratio/mean": 0.9999959468841553, |
| "sampling/importance_sampling_ratio/min": 0.7060609459877014, |
| "sampling/sampling_logp_difference/max": 0.3530550003051758, |
| "sampling/sampling_logp_difference/mean": 0.011036572977900505, |
| "step": 43 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15865.0, |
| "completions/mean_length": 12600.375, |
| "completions/mean_terminated_length": 12059.857421875, |
| "completions/min_length": 5038.0, |
| "completions/min_terminated_length": 5038.0, |
| "entropy": 0.29098295606672764, |
| "epoch": 0.002529026324864927, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2217897.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997394680976868, |
| "sampling/importance_sampling_ratio/min": 0.047701530158519745, |
| "sampling/sampling_logp_difference/max": 3.0427918434143066, |
| "sampling/sampling_logp_difference/mean": 0.012049276381731033, |
| "step": 44 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5978.0, |
| "completions/max_terminated_length": 5978.0, |
| "completions/mean_length": 5043.375, |
| "completions/mean_terminated_length": 5043.375, |
| "completions/min_length": 3933.0, |
| "completions/min_terminated_length": 3933.0, |
| "entropy": 0.4449828788638115, |
| "epoch": 0.0025865041958845845, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2259348.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.769199252128601, |
| "sampling/importance_sampling_ratio/mean": 0.9998717904090881, |
| "sampling/importance_sampling_ratio/min": 0.5246632695198059, |
| "sampling/sampling_logp_difference/max": 0.6449985504150391, |
| "sampling/sampling_logp_difference/mean": 0.017409298568964005, |
| "step": 45 |
| }, |
| { |
| "clip_ratio/high_max": 4.307374183554202e-05, |
| "clip_ratio/high_mean": 4.307374183554202e-05, |
| "clip_ratio/low_mean": 0.0003977762389695272, |
| "clip_ratio/low_min": 0.0003977762389695272, |
| "clip_ratio/region_mean": 0.0004408499808050692, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2902.0, |
| "completions/max_terminated_length": 2902.0, |
| "completions/mean_length": 1450.25, |
| "completions/mean_terminated_length": 1450.25, |
| "completions/min_length": 557.0, |
| "completions/min_terminated_length": 557.0, |
| "entropy": 0.2576371170580387, |
| "epoch": 0.0026439820669042417, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.037199344485998154, |
| "learning_rate": 1e-05, |
| "loss": -0.1005, |
| "num_tokens": 2271630.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.4901281595230103, |
| "sampling/importance_sampling_ratio/mean": 0.9999934434890747, |
| "sampling/importance_sampling_ratio/min": 0.6777300834655762, |
| "sampling/sampling_logp_difference/max": 0.3988621234893799, |
| "sampling/sampling_logp_difference/mean": 0.011057121679186821, |
| "step": 46 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00010794473200803623, |
| "clip_ratio/low_min": 0.00010794473200803623, |
| "clip_ratio/region_mean": 0.00010794473200803623, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2922.0, |
| "completions/max_terminated_length": 2922.0, |
| "completions/mean_length": 1511.75, |
| "completions/mean_terminated_length": 1511.75, |
| "completions/min_length": 729.0, |
| "completions/min_terminated_length": 729.0, |
| "entropy": 0.5865042172372341, |
| "epoch": 0.0027014599379238994, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02290233224630356, |
| "learning_rate": 1e-05, |
| "loss": -0.1436, |
| "num_tokens": 2284788.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.3138504028320312, |
| "sampling/importance_sampling_ratio/mean": 1.0002350807189941, |
| "sampling/importance_sampling_ratio/min": 0.6800664067268372, |
| "sampling/sampling_logp_difference/max": 0.38556480407714844, |
| "sampling/sampling_logp_difference/mean": 0.016870463266968727, |
| "step": 47 |
| }, |
| { |
| "clip_ratio/high_max": 3.955070496886037e-05, |
| "clip_ratio/high_mean": 3.955070496886037e-05, |
| "clip_ratio/low_mean": 2.347930785617791e-05, |
| "clip_ratio/low_min": 2.347930785617791e-05, |
| "clip_ratio/region_mean": 6.303001282503828e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10707.0, |
| "completions/max_terminated_length": 10707.0, |
| "completions/mean_length": 6775.0, |
| "completions/mean_terminated_length": 6775.0, |
| "completions/min_length": 1813.0, |
| "completions/min_terminated_length": 1813.0, |
| "entropy": 0.6008276157081127, |
| "epoch": 0.0027589378089435566, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005901714321225882, |
| "learning_rate": 1e-05, |
| "loss": 0.3085, |
| "num_tokens": 2340020.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998616576194763, |
| "sampling/importance_sampling_ratio/min": 0.5181363224983215, |
| "sampling/sampling_logp_difference/max": 0.7540040016174316, |
| "sampling/sampling_logp_difference/mean": 0.018575357273221016, |
| "step": 48 |
| }, |
| { |
| "clip_ratio/high_max": 7.636024292878574e-05, |
| "clip_ratio/high_mean": 7.636024292878574e-05, |
| "clip_ratio/low_mean": 0.00025442260812269524, |
| "clip_ratio/low_min": 0.00025442260812269524, |
| "clip_ratio/region_mean": 0.000330782851051481, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14986.0, |
| "completions/mean_length": 11027.875, |
| "completions/mean_terminated_length": 9242.5, |
| "completions/min_length": 2475.0, |
| "completions/min_terminated_length": 2475.0, |
| "entropy": 0.45021577924489975, |
| "epoch": 0.0028164156799632143, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021097583696246147, |
| "learning_rate": 1e-05, |
| "loss": 0.0651, |
| "num_tokens": 2429131.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999630451202393, |
| "sampling/importance_sampling_ratio/min": 0.4296703040599823, |
| "sampling/sampling_logp_difference/max": 0.8447370529174805, |
| "sampling/sampling_logp_difference/mean": 0.020308000966906548, |
| "step": 49 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10777.0, |
| "completions/max_terminated_length": 10777.0, |
| "completions/mean_length": 8693.625, |
| "completions/mean_terminated_length": 8693.625, |
| "completions/min_length": 5111.0, |
| "completions/min_terminated_length": 5111.0, |
| "entropy": 0.6992930024862289, |
| "epoch": 0.0028738935509828715, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2500248.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998568892478943, |
| "sampling/importance_sampling_ratio/min": 0.37074294686317444, |
| "sampling/sampling_logp_difference/max": 0.9922463893890381, |
| "sampling/sampling_logp_difference/mean": 0.027893485501408577, |
| "step": 50 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00013150973245501518, |
| "clip_ratio/low_min": 0.00013150973245501518, |
| "clip_ratio/region_mean": 0.00013150973245501518, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3448.0, |
| "completions/max_terminated_length": 3448.0, |
| "completions/mean_length": 2144.5, |
| "completions/mean_terminated_length": 2144.5, |
| "completions/min_length": 1325.0, |
| "completions/min_terminated_length": 1325.0, |
| "entropy": 0.20928144082427025, |
| "epoch": 0.002931371422002529, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05555763095617294, |
| "learning_rate": 1e-05, |
| "loss": -0.0964, |
| "num_tokens": 2518124.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4257904291152954, |
| "sampling/importance_sampling_ratio/mean": 0.9999476671218872, |
| "sampling/importance_sampling_ratio/min": 0.6867164969444275, |
| "sampling/sampling_logp_difference/max": 0.37583374977111816, |
| "sampling/sampling_logp_difference/mean": 0.00927009154111147, |
| "step": 51 |
| }, |
| { |
| "clip_ratio/high_max": 6.471852248068899e-05, |
| "clip_ratio/high_mean": 6.471852248068899e-05, |
| "clip_ratio/low_mean": 6.35001269984059e-05, |
| "clip_ratio/low_min": 6.35001269984059e-05, |
| "clip_ratio/region_mean": 0.0001282186494790949, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9808.0, |
| "completions/max_terminated_length": 9808.0, |
| "completions/mean_length": 6485.875, |
| "completions/mean_terminated_length": 6485.875, |
| "completions/min_length": 4964.0, |
| "completions/min_terminated_length": 4964.0, |
| "entropy": 0.26626385375857353, |
| "epoch": 0.0029888492930221864, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05183282867074013, |
| "learning_rate": 1e-05, |
| "loss": 0.0755, |
| "num_tokens": 2571451.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000094175338745, |
| "sampling/importance_sampling_ratio/min": 0.46055662631988525, |
| "sampling/sampling_logp_difference/max": 0.9583044052124023, |
| "sampling/sampling_logp_difference/mean": 0.011562701314687729, |
| "step": 52 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5567.0, |
| "completions/max_terminated_length": 5567.0, |
| "completions/mean_length": 4548.25, |
| "completions/mean_terminated_length": 4548.25, |
| "completions/min_length": 2948.0, |
| "completions/min_terminated_length": 2948.0, |
| "entropy": 0.3451690226793289, |
| "epoch": 0.003046327164041844, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2608957.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4250214099884033, |
| "sampling/importance_sampling_ratio/mean": 1.0001447200775146, |
| "sampling/importance_sampling_ratio/min": 0.6228136420249939, |
| "sampling/sampling_logp_difference/max": 0.4735078811645508, |
| "sampling/sampling_logp_difference/mean": 0.013225145637989044, |
| "step": 53 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002130183820554521, |
| "clip_ratio/high_mean": 0.0002130183820554521, |
| "clip_ratio/low_mean": 0.000633551215287298, |
| "clip_ratio/low_min": 0.000633551215287298, |
| "clip_ratio/region_mean": 0.0008465695973427501, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8079.0, |
| "completions/max_terminated_length": 8079.0, |
| "completions/mean_length": 5219.375, |
| "completions/mean_terminated_length": 5219.375, |
| "completions/min_length": 2274.0, |
| "completions/min_terminated_length": 2274.0, |
| "entropy": 0.681399904191494, |
| "epoch": 0.0031038050350615013, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.038597557693719864, |
| "learning_rate": 1e-05, |
| "loss": -0.2008, |
| "num_tokens": 2653176.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000501871109009, |
| "sampling/importance_sampling_ratio/min": 0.5542822480201721, |
| "sampling/sampling_logp_difference/max": 0.8316373825073242, |
| "sampling/sampling_logp_difference/mean": 0.02502310648560524, |
| "step": 54 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13798.0, |
| "completions/mean_length": 12660.875, |
| "completions/mean_terminated_length": 11419.833984375, |
| "completions/min_length": 9486.0, |
| "completions/min_terminated_length": 9486.0, |
| "entropy": 0.6816116347908974, |
| "epoch": 0.003161282906081159, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2755687.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000107288360596, |
| "sampling/importance_sampling_ratio/min": 0.04938606917858124, |
| "sampling/sampling_logp_difference/max": 3.008086919784546, |
| "sampling/sampling_logp_difference/mean": 0.026097681373357773, |
| "step": 55 |
| }, |
| { |
| "clip_ratio/high_max": 0.00016612962645012885, |
| "clip_ratio/high_mean": 0.00016612962645012885, |
| "clip_ratio/low_mean": 0.00011158721463289112, |
| "clip_ratio/low_min": 0.00011158721463289112, |
| "clip_ratio/region_mean": 0.00027771684108301997, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6250.0, |
| "completions/max_terminated_length": 6250.0, |
| "completions/mean_length": 5129.125, |
| "completions/mean_terminated_length": 5129.125, |
| "completions/min_length": 3532.0, |
| "completions/min_terminated_length": 3532.0, |
| "entropy": 0.3505010213702917, |
| "epoch": 0.003218760777100816, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009208712726831436, |
| "learning_rate": 1e-05, |
| "loss": 0.0324, |
| "num_tokens": 2797536.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4510549306869507, |
| "sampling/importance_sampling_ratio/mean": 0.9997953176498413, |
| "sampling/importance_sampling_ratio/min": 0.4308338165283203, |
| "sampling/sampling_logp_difference/max": 0.8420329093933105, |
| "sampling/sampling_logp_difference/mean": 0.012967368587851524, |
| "step": 56 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12147.0, |
| "completions/max_terminated_length": 12147.0, |
| "completions/mean_length": 7532.125, |
| "completions/mean_terminated_length": 7532.125, |
| "completions/min_length": 3595.0, |
| "completions/min_terminated_length": 3595.0, |
| "entropy": 0.545828327536583, |
| "epoch": 0.0032762386481204734, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2859137.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998393058776855, |
| "sampling/importance_sampling_ratio/min": 0.3288186490535736, |
| "sampling/sampling_logp_difference/max": 1.2061529159545898, |
| "sampling/sampling_logp_difference/mean": 0.02107314020395279, |
| "step": 57 |
| }, |
| { |
| "clip_ratio/high_max": 5.484861685545184e-05, |
| "clip_ratio/high_mean": 5.484861685545184e-05, |
| "clip_ratio/low_mean": 8.833922038320452e-05, |
| "clip_ratio/low_min": 8.833922038320452e-05, |
| "clip_ratio/region_mean": 0.00014318783723865636, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2403.0, |
| "completions/max_terminated_length": 2403.0, |
| "completions/mean_length": 1772.375, |
| "completions/mean_terminated_length": 1772.375, |
| "completions/min_length": 1263.0, |
| "completions/min_terminated_length": 1263.0, |
| "entropy": 0.2984078824520111, |
| "epoch": 0.003333716519140131, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01238580048084259, |
| "learning_rate": 1e-05, |
| "loss": -0.0716, |
| "num_tokens": 2874292.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3563395738601685, |
| "sampling/importance_sampling_ratio/mean": 1.0003012418746948, |
| "sampling/importance_sampling_ratio/min": 0.6631757616996765, |
| "sampling/sampling_logp_difference/max": 0.4107152223587036, |
| "sampling/sampling_logp_difference/mean": 0.011635399423539639, |
| "step": 58 |
| }, |
| { |
| "clip_ratio/high_max": 9.920635056914762e-05, |
| "clip_ratio/high_mean": 9.920635056914762e-05, |
| "clip_ratio/low_mean": 0.00020814879826502874, |
| "clip_ratio/low_min": 0.00020814879826502874, |
| "clip_ratio/region_mean": 0.00030735514883417636, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1637.0, |
| "completions/max_terminated_length": 1637.0, |
| "completions/mean_length": 1393.875, |
| "completions/mean_terminated_length": 1393.875, |
| "completions/min_length": 1012.0, |
| "completions/min_terminated_length": 1012.0, |
| "entropy": 0.24145445227622986, |
| "epoch": 0.0033911943901597883, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021861892193555832, |
| "learning_rate": 1e-05, |
| "loss": -0.0578, |
| "num_tokens": 2887627.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4735182523727417, |
| "sampling/importance_sampling_ratio/mean": 1.0001130104064941, |
| "sampling/importance_sampling_ratio/min": 0.631110429763794, |
| "sampling/sampling_logp_difference/max": 0.46027445793151855, |
| "sampling/sampling_logp_difference/mean": 0.009926659055054188, |
| "step": 59 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6529.0, |
| "completions/max_terminated_length": 6529.0, |
| "completions/mean_length": 4031.625, |
| "completions/mean_terminated_length": 4031.625, |
| "completions/min_length": 1792.0, |
| "completions/min_terminated_length": 1792.0, |
| "entropy": 0.2522127367556095, |
| "epoch": 0.003448672261179446, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 2920792.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5025330781936646, |
| "sampling/importance_sampling_ratio/mean": 1.0001198053359985, |
| "sampling/importance_sampling_ratio/min": 0.6018195152282715, |
| "sampling/sampling_logp_difference/max": 0.5077977180480957, |
| "sampling/sampling_logp_difference/mean": 0.010761220008134842, |
| "step": 60 |
| }, |
| { |
| "clip_ratio/high_max": 0.00024247844339697622, |
| "clip_ratio/high_mean": 0.00024247844339697622, |
| "clip_ratio/low_mean": 4.752851600642316e-05, |
| "clip_ratio/low_min": 4.752851600642316e-05, |
| "clip_ratio/region_mean": 0.0002900069594033994, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3546.0, |
| "completions/max_terminated_length": 3546.0, |
| "completions/mean_length": 2080.25, |
| "completions/mean_terminated_length": 2080.25, |
| "completions/min_length": 967.0, |
| "completions/min_terminated_length": 967.0, |
| "entropy": 0.2911796625703573, |
| "epoch": 0.003506150132199103, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023584146052598953, |
| "learning_rate": 1e-05, |
| "loss": -0.0732, |
| "num_tokens": 2938794.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4307900667190552, |
| "sampling/importance_sampling_ratio/mean": 0.9999948740005493, |
| "sampling/importance_sampling_ratio/min": 0.69541996717453, |
| "sampling/sampling_logp_difference/max": 0.3632392883300781, |
| "sampling/sampling_logp_difference/mean": 0.012206352315843105, |
| "step": 61 |
| }, |
| { |
| "clip_ratio/high_max": 1.4455880773311947e-05, |
| "clip_ratio/high_mean": 1.4455880773311947e-05, |
| "clip_ratio/low_mean": 0.000513993112690514, |
| "clip_ratio/low_min": 0.000513993112690514, |
| "clip_ratio/region_mean": 0.0005284489934638259, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8647.0, |
| "completions/max_terminated_length": 8647.0, |
| "completions/mean_length": 4010.375, |
| "completions/mean_terminated_length": 4010.375, |
| "completions/min_length": 740.0, |
| "completions/min_terminated_length": 740.0, |
| "entropy": 0.4742611162364483, |
| "epoch": 0.003563628003218761, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03585368022322655, |
| "learning_rate": 1e-05, |
| "loss": -0.0353, |
| "num_tokens": 2975653.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6222310066223145, |
| "sampling/importance_sampling_ratio/mean": 1.0001109838485718, |
| "sampling/importance_sampling_ratio/min": 0.608767569065094, |
| "sampling/sampling_logp_difference/max": 0.4963188171386719, |
| "sampling/sampling_logp_difference/mean": 0.020225893706083298, |
| "step": 62 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017478327390563209, |
| "clip_ratio/high_mean": 0.00017478327390563209, |
| "clip_ratio/low_mean": 0.00015009606431704015, |
| "clip_ratio/low_min": 0.00015009606431704015, |
| "clip_ratio/region_mean": 0.00032487933822267223, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8328.0, |
| "completions/max_terminated_length": 8328.0, |
| "completions/mean_length": 6091.625, |
| "completions/mean_terminated_length": 6091.625, |
| "completions/min_length": 3429.0, |
| "completions/min_terminated_length": 3429.0, |
| "entropy": 0.4262738786637783, |
| "epoch": 0.003621105874238418, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0088336281478405, |
| "learning_rate": 1e-05, |
| "loss": 0.1299, |
| "num_tokens": 3025186.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6910793781280518, |
| "sampling/importance_sampling_ratio/mean": 1.0002009868621826, |
| "sampling/importance_sampling_ratio/min": 0.6178554892539978, |
| "sampling/sampling_logp_difference/max": 0.525367021560669, |
| "sampling/sampling_logp_difference/mean": 0.016626553609967232, |
| "step": 63 |
| }, |
| { |
| "clip_ratio/high_max": 8.472792978864163e-05, |
| "clip_ratio/high_mean": 8.472792978864163e-05, |
| "clip_ratio/low_mean": 0.0005569919776462484, |
| "clip_ratio/low_min": 0.0005569919776462484, |
| "clip_ratio/region_mean": 0.00064171990743489, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12677.0, |
| "completions/mean_length": 9709.75, |
| "completions/mean_terminated_length": 8756.2861328125, |
| "completions/min_length": 4627.0, |
| "completions/min_terminated_length": 4627.0, |
| "entropy": 0.5099417231976986, |
| "epoch": 0.0036785837452580758, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025481535121798515, |
| "learning_rate": 1e-05, |
| "loss": 0.057, |
| "num_tokens": 3104288.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999805212020874, |
| "sampling/importance_sampling_ratio/min": 0.3367921710014343, |
| "sampling/sampling_logp_difference/max": 1.3402471542358398, |
| "sampling/sampling_logp_difference/mean": 0.021615201607346535, |
| "step": 64 |
| }, |
| { |
| "clip_ratio/high_max": 6.347670478135115e-05, |
| "clip_ratio/high_mean": 6.347670478135115e-05, |
| "clip_ratio/low_mean": 0.0005828946596011519, |
| "clip_ratio/low_min": 0.0005828946596011519, |
| "clip_ratio/region_mean": 0.0006463713643825031, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14769.0, |
| "completions/max_terminated_length": 14769.0, |
| "completions/mean_length": 11125.625, |
| "completions/mean_terminated_length": 11125.625, |
| "completions/min_length": 8889.0, |
| "completions/min_terminated_length": 8889.0, |
| "entropy": 0.5693316161632538, |
| "epoch": 0.003736061616277733, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013858629390597343, |
| "learning_rate": 1e-05, |
| "loss": 0.1261, |
| "num_tokens": 3194197.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999197721481323, |
| "sampling/importance_sampling_ratio/min": 0.20770741999149323, |
| "sampling/sampling_logp_difference/max": 1.571624755859375, |
| "sampling/sampling_logp_difference/mean": 0.022273730486631393, |
| "step": 65 |
| }, |
| { |
| "clip_ratio/high_max": 4.7023926526890136e-05, |
| "clip_ratio/high_mean": 4.7023926526890136e-05, |
| "clip_ratio/low_mean": 0.0003786109300563112, |
| "clip_ratio/low_min": 0.0003786109300563112, |
| "clip_ratio/region_mean": 0.0004256348565832013, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7961.0, |
| "completions/max_terminated_length": 7961.0, |
| "completions/mean_length": 5469.125, |
| "completions/mean_terminated_length": 5469.125, |
| "completions/min_length": 3050.0, |
| "completions/min_terminated_length": 3050.0, |
| "entropy": 0.37077387794852257, |
| "epoch": 0.0037935394872973907, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014726302586495876, |
| "learning_rate": 1e-05, |
| "loss": -0.0125, |
| "num_tokens": 3239414.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4613500833511353, |
| "sampling/importance_sampling_ratio/mean": 1.00039803981781, |
| "sampling/importance_sampling_ratio/min": 0.6130610108375549, |
| "sampling/sampling_logp_difference/max": 0.48929083347320557, |
| "sampling/sampling_logp_difference/mean": 0.015297925099730492, |
| "step": 66 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014758230099687353, |
| "clip_ratio/high_mean": 0.00014758230099687353, |
| "clip_ratio/low_mean": 0.00034344276355113834, |
| "clip_ratio/low_min": 0.00034344276355113834, |
| "clip_ratio/region_mean": 0.0004910250645480119, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15818.0, |
| "completions/max_terminated_length": 15818.0, |
| "completions/mean_length": 4518.875, |
| "completions/mean_terminated_length": 4518.875, |
| "completions/min_length": 710.0, |
| "completions/min_terminated_length": 710.0, |
| "entropy": 0.46321990713477135, |
| "epoch": 0.003851017358317048, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.052609484642744064, |
| "learning_rate": 1e-05, |
| "loss": 0.1724, |
| "num_tokens": 3276549.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999647736549377, |
| "sampling/importance_sampling_ratio/min": 0.5366171598434448, |
| "sampling/sampling_logp_difference/max": 0.7253210544586182, |
| "sampling/sampling_logp_difference/mean": 0.019297108054161072, |
| "step": 67 |
| }, |
| { |
| "clip_ratio/high_max": 5.84329609409906e-05, |
| "clip_ratio/high_mean": 5.84329609409906e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.84329609409906e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6515.0, |
| "completions/max_terminated_length": 6515.0, |
| "completions/mean_length": 3167.125, |
| "completions/mean_terminated_length": 3167.125, |
| "completions/min_length": 930.0, |
| "completions/min_terminated_length": 930.0, |
| "entropy": 0.3499685563147068, |
| "epoch": 0.0039084952293367056, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008895262144505978, |
| "learning_rate": 1e-05, |
| "loss": -0.2496, |
| "num_tokens": 3303014.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.476549744606018, |
| "sampling/importance_sampling_ratio/mean": 1.000215768814087, |
| "sampling/importance_sampling_ratio/min": 0.6207373738288879, |
| "sampling/sampling_logp_difference/max": 0.47684717178344727, |
| "sampling/sampling_logp_difference/mean": 0.015071067027747631, |
| "step": 68 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.4886269127600826e-05, |
| "clip_ratio/low_min": 1.4886269127600826e-05, |
| "clip_ratio/region_mean": 1.4886269127600826e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8397.0, |
| "completions/max_terminated_length": 8397.0, |
| "completions/mean_length": 4625.375, |
| "completions/mean_terminated_length": 4625.375, |
| "completions/min_length": 1610.0, |
| "completions/min_terminated_length": 1610.0, |
| "entropy": 0.5824310332536697, |
| "epoch": 0.003965973100356363, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.1078847348690033, |
| "learning_rate": 1e-05, |
| "loss": 0.2885, |
| "num_tokens": 3341649.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6486999988555908, |
| "sampling/importance_sampling_ratio/mean": 0.9998449683189392, |
| "sampling/importance_sampling_ratio/min": 0.6175811886787415, |
| "sampling/sampling_logp_difference/max": 0.4999871253967285, |
| "sampling/sampling_logp_difference/mean": 0.017474355176091194, |
| "step": 69 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3218.0, |
| "completions/max_terminated_length": 3218.0, |
| "completions/mean_length": 1611.0, |
| "completions/mean_terminated_length": 1611.0, |
| "completions/min_length": 498.0, |
| "completions/min_terminated_length": 498.0, |
| "entropy": 0.44486694410443306, |
| "epoch": 0.00402345097137602, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 3355417.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2839782238006592, |
| "sampling/importance_sampling_ratio/mean": 1.0003026723861694, |
| "sampling/importance_sampling_ratio/min": 0.6977416276931763, |
| "sampling/sampling_logp_difference/max": 0.3599064350128174, |
| "sampling/sampling_logp_difference/mean": 0.016955537721514702, |
| "step": 70 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11105.0, |
| "completions/max_terminated_length": 11105.0, |
| "completions/mean_length": 6453.75, |
| "completions/mean_terminated_length": 6453.75, |
| "completions/min_length": 2829.0, |
| "completions/min_terminated_length": 2829.0, |
| "entropy": 0.5774021446704865, |
| "epoch": 0.004080928842395678, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 3408039.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9741193056106567, |
| "sampling/importance_sampling_ratio/mean": 1.0002410411834717, |
| "sampling/importance_sampling_ratio/min": 0.5421406626701355, |
| "sampling/sampling_logp_difference/max": 0.6801223754882812, |
| "sampling/sampling_logp_difference/mean": 0.017538031563162804, |
| "step": 71 |
| }, |
| { |
| "clip_ratio/high_max": 7.407679731841199e-05, |
| "clip_ratio/high_mean": 7.407679731841199e-05, |
| "clip_ratio/low_mean": 0.0003438445564825088, |
| "clip_ratio/low_min": 0.0003438445564825088, |
| "clip_ratio/region_mean": 0.00041792135380092077, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9270.0, |
| "completions/max_terminated_length": 9270.0, |
| "completions/mean_length": 5584.875, |
| "completions/mean_terminated_length": 5584.875, |
| "completions/min_length": 498.0, |
| "completions/min_terminated_length": 498.0, |
| "entropy": 0.44911012426018715, |
| "epoch": 0.004138406713415335, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016307076439261436, |
| "learning_rate": 1e-05, |
| "loss": 0.2551, |
| "num_tokens": 3454102.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8314783573150635, |
| "sampling/importance_sampling_ratio/mean": 1.0001745223999023, |
| "sampling/importance_sampling_ratio/min": 0.47552651166915894, |
| "sampling/sampling_logp_difference/max": 0.7433326244354248, |
| "sampling/sampling_logp_difference/mean": 0.01905350759625435, |
| "step": 72 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010738677883637138, |
| "clip_ratio/high_mean": 0.00010738677883637138, |
| "clip_ratio/low_mean": 0.0003271719397162087, |
| "clip_ratio/low_min": 0.0003271719397162087, |
| "clip_ratio/region_mean": 0.0004345587185525801, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11830.0, |
| "completions/max_terminated_length": 11830.0, |
| "completions/mean_length": 6688.75, |
| "completions/mean_terminated_length": 6688.75, |
| "completions/min_length": 2061.0, |
| "completions/min_terminated_length": 2061.0, |
| "entropy": 0.7740826979279518, |
| "epoch": 0.004195884584434992, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.029672987759113312, |
| "learning_rate": 1e-05, |
| "loss": 0.2613, |
| "num_tokens": 3508732.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.8884772062301636, |
| "sampling/importance_sampling_ratio/mean": 1.000227689743042, |
| "sampling/importance_sampling_ratio/min": 0.4901808798313141, |
| "sampling/sampling_logp_difference/max": 0.7129808664321899, |
| "sampling/sampling_logp_difference/mean": 0.02653191238641739, |
| "step": 73 |
| }, |
| { |
| "clip_ratio/high_max": 6.572279562533367e-05, |
| "clip_ratio/high_mean": 6.572279562533367e-05, |
| "clip_ratio/low_mean": 0.0006653135860688053, |
| "clip_ratio/low_min": 0.0006653135860688053, |
| "clip_ratio/region_mean": 0.000731036381694139, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14942.0, |
| "completions/mean_length": 11513.625, |
| "completions/mean_terminated_length": 10817.857421875, |
| "completions/min_length": 6219.0, |
| "completions/min_terminated_length": 6219.0, |
| "entropy": 0.6061702743172646, |
| "epoch": 0.00425336245545465, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023116735741496086, |
| "learning_rate": 1e-05, |
| "loss": 0.1467, |
| "num_tokens": 3601689.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999862790107727, |
| "sampling/importance_sampling_ratio/min": 0.0003409076016396284, |
| "sampling/sampling_logp_difference/max": 7.983899116516113, |
| "sampling/sampling_logp_difference/mean": 0.022893456742167473, |
| "step": 74 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3982.0, |
| "completions/max_terminated_length": 3982.0, |
| "completions/mean_length": 2700.875, |
| "completions/mean_terminated_length": 2700.875, |
| "completions/min_length": 659.0, |
| "completions/min_terminated_length": 659.0, |
| "entropy": 0.1945251878350973, |
| "epoch": 0.0043108403264743075, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 3624448.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4087806940078735, |
| "sampling/importance_sampling_ratio/mean": 0.9997677803039551, |
| "sampling/importance_sampling_ratio/min": 0.44225573539733887, |
| "sampling/sampling_logp_difference/max": 0.8158669471740723, |
| "sampling/sampling_logp_difference/mean": 0.007831585593521595, |
| "step": 75 |
| }, |
| { |
| "clip_ratio/high_max": 2.9169259505579248e-05, |
| "clip_ratio/high_mean": 2.9169259505579248e-05, |
| "clip_ratio/low_mean": 0.0010002476192312315, |
| "clip_ratio/low_min": 0.0010002476192312315, |
| "clip_ratio/region_mean": 0.0010294168787368108, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12856.0, |
| "completions/max_terminated_length": 12856.0, |
| "completions/mean_length": 9036.75, |
| "completions/mean_terminated_length": 9036.75, |
| "completions/min_length": 3947.0, |
| "completions/min_terminated_length": 3947.0, |
| "entropy": 0.584886621683836, |
| "epoch": 0.004368318197493965, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007610360626131296, |
| "learning_rate": 1e-05, |
| "loss": -0.1496, |
| "num_tokens": 3698494.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000782012939453, |
| "sampling/importance_sampling_ratio/min": 0.4218885898590088, |
| "sampling/sampling_logp_difference/max": 0.8630140423774719, |
| "sampling/sampling_logp_difference/mean": 0.02367301844060421, |
| "step": 76 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 5.2698145736940205e-05, |
| "clip_ratio/low_min": 5.2698145736940205e-05, |
| "clip_ratio/region_mean": 5.2698145736940205e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2372.0, |
| "completions/max_terminated_length": 2372.0, |
| "completions/mean_length": 1625.625, |
| "completions/mean_terminated_length": 1625.625, |
| "completions/min_length": 835.0, |
| "completions/min_terminated_length": 835.0, |
| "entropy": 0.7885649353265762, |
| "epoch": 0.004425796068513622, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0428813137114048, |
| "learning_rate": 1e-05, |
| "loss": -0.0962, |
| "num_tokens": 3712811.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.2535712718963623, |
| "sampling/importance_sampling_ratio/mean": 0.9997212886810303, |
| "sampling/importance_sampling_ratio/min": 0.6434804797172546, |
| "sampling/sampling_logp_difference/max": 0.44086360931396484, |
| "sampling/sampling_logp_difference/mean": 0.020571956411004066, |
| "step": 77 |
| }, |
| { |
| "clip_ratio/high_max": 3.8287769712042063e-05, |
| "clip_ratio/high_mean": 3.8287769712042063e-05, |
| "clip_ratio/low_mean": 0.0008040851753321476, |
| "clip_ratio/low_min": 0.0008040851753321476, |
| "clip_ratio/region_mean": 0.0008423729450441897, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13059.0, |
| "completions/mean_length": 11664.875, |
| "completions/mean_terminated_length": 10091.833984375, |
| "completions/min_length": 6610.0, |
| "completions/min_terminated_length": 6610.0, |
| "entropy": 0.5955601409077644, |
| "epoch": 0.00448327393953328, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009304461069405079, |
| "learning_rate": 1e-05, |
| "loss": -0.0425, |
| "num_tokens": 3807858.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000051736831665, |
| "sampling/importance_sampling_ratio/min": 0.40974199771881104, |
| "sampling/sampling_logp_difference/max": 0.8922276496887207, |
| "sampling/sampling_logp_difference/mean": 0.02575019933283329, |
| "step": 78 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8903.0, |
| "completions/max_terminated_length": 8903.0, |
| "completions/mean_length": 4943.375, |
| "completions/mean_terminated_length": 4943.375, |
| "completions/min_length": 2102.0, |
| "completions/min_terminated_length": 2102.0, |
| "entropy": 0.3662714660167694, |
| "epoch": 0.004540751810552937, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 3849109.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7069648504257202, |
| "sampling/importance_sampling_ratio/mean": 1.0002577304840088, |
| "sampling/importance_sampling_ratio/min": 0.42224836349487305, |
| "sampling/sampling_logp_difference/max": 0.8621616363525391, |
| "sampling/sampling_logp_difference/mean": 0.014560350216925144, |
| "step": 79 |
| }, |
| { |
| "clip_ratio/high_max": 4.0380606151302345e-05, |
| "clip_ratio/high_mean": 4.0380606151302345e-05, |
| "clip_ratio/low_mean": 0.00026428982164361514, |
| "clip_ratio/low_min": 0.00026428982164361514, |
| "clip_ratio/region_mean": 0.0003046704277949175, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8607.0, |
| "completions/max_terminated_length": 8607.0, |
| "completions/mean_length": 5977.625, |
| "completions/mean_terminated_length": 5977.625, |
| "completions/min_length": 3968.0, |
| "completions/min_terminated_length": 3968.0, |
| "entropy": 0.48602984473109245, |
| "epoch": 0.004598229681572595, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02175828441977501, |
| "learning_rate": 1e-05, |
| "loss": 0.047, |
| "num_tokens": 3898130.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6929270029067993, |
| "sampling/importance_sampling_ratio/mean": 0.999931812286377, |
| "sampling/importance_sampling_ratio/min": 0.5149385333061218, |
| "sampling/sampling_logp_difference/max": 0.6637077331542969, |
| "sampling/sampling_logp_difference/mean": 0.015433047898113728, |
| "step": 80 |
| }, |
| { |
| "clip_ratio/high_max": 3.3530042856000364e-05, |
| "clip_ratio/high_mean": 3.3530042856000364e-05, |
| "clip_ratio/low_mean": 0.00010602205293253064, |
| "clip_ratio/low_min": 0.00010602205293253064, |
| "clip_ratio/region_mean": 0.000139552095788531, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3728.0, |
| "completions/max_terminated_length": 3728.0, |
| "completions/mean_length": 2809.5, |
| "completions/mean_terminated_length": 2809.5, |
| "completions/min_length": 2034.0, |
| "completions/min_terminated_length": 2034.0, |
| "entropy": 0.289394024759531, |
| "epoch": 0.004655707552592252, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011299347504973412, |
| "learning_rate": 1e-05, |
| "loss": -0.0571, |
| "num_tokens": 3921782.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3607003688812256, |
| "sampling/importance_sampling_ratio/mean": 1.000383973121643, |
| "sampling/importance_sampling_ratio/min": 0.5568434596061707, |
| "sampling/sampling_logp_difference/max": 0.5854711532592773, |
| "sampling/sampling_logp_difference/mean": 0.010291689075529575, |
| "step": 81 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010765574734250549, |
| "clip_ratio/high_mean": 0.00010765574734250549, |
| "clip_ratio/low_mean": 0.0003638860216597095, |
| "clip_ratio/low_min": 0.0003638860216597095, |
| "clip_ratio/region_mean": 0.000471541769002215, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5725.0, |
| "completions/max_terminated_length": 5725.0, |
| "completions/mean_length": 4045.5, |
| "completions/mean_terminated_length": 4045.5, |
| "completions/min_length": 2295.0, |
| "completions/min_terminated_length": 2295.0, |
| "entropy": 0.5646822117269039, |
| "epoch": 0.004713185423611909, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023033710196614265, |
| "learning_rate": 1e-05, |
| "loss": -0.0664, |
| "num_tokens": 3954906.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.473975658416748, |
| "sampling/importance_sampling_ratio/mean": 0.9994459748268127, |
| "sampling/importance_sampling_ratio/min": 0.667750358581543, |
| "sampling/sampling_logp_difference/max": 0.40384089946746826, |
| "sampling/sampling_logp_difference/mean": 0.02020481415092945, |
| "step": 82 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9426.0, |
| "completions/max_terminated_length": 9426.0, |
| "completions/mean_length": 5180.5, |
| "completions/mean_terminated_length": 5180.5, |
| "completions/min_length": 3161.0, |
| "completions/min_terminated_length": 3161.0, |
| "entropy": 0.5163888968527317, |
| "epoch": 0.004770663294631567, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 3997582.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5550614595413208, |
| "sampling/importance_sampling_ratio/mean": 0.9998461604118347, |
| "sampling/importance_sampling_ratio/min": 0.41633352637290955, |
| "sampling/sampling_logp_difference/max": 0.8762686252593994, |
| "sampling/sampling_logp_difference/mean": 0.01981530524790287, |
| "step": 83 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4797.0, |
| "completions/max_terminated_length": 4797.0, |
| "completions/mean_length": 3375.0, |
| "completions/mean_terminated_length": 3375.0, |
| "completions/min_length": 1937.0, |
| "completions/min_terminated_length": 1937.0, |
| "entropy": 0.23298371024429798, |
| "epoch": 0.004828141165651225, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 4025438.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.409183382987976, |
| "sampling/importance_sampling_ratio/mean": 1.0002565383911133, |
| "sampling/importance_sampling_ratio/min": 0.670382022857666, |
| "sampling/sampling_logp_difference/max": 0.39990758895874023, |
| "sampling/sampling_logp_difference/mean": 0.008978866040706635, |
| "step": 84 |
| }, |
| { |
| "clip_ratio/high_max": 6.28058460279135e-05, |
| "clip_ratio/high_mean": 6.28058460279135e-05, |
| "clip_ratio/low_mean": 0.0004359620616014581, |
| "clip_ratio/low_min": 0.0004359620616014581, |
| "clip_ratio/region_mean": 0.0004987679076293716, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13690.0, |
| "completions/max_terminated_length": 13690.0, |
| "completions/mean_length": 9661.75, |
| "completions/mean_terminated_length": 9661.75, |
| "completions/min_length": 7069.0, |
| "completions/min_terminated_length": 7069.0, |
| "entropy": 0.41262468323111534, |
| "epoch": 0.0048856190366708815, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013142373412847519, |
| "learning_rate": 1e-05, |
| "loss": -0.0268, |
| "num_tokens": 4103852.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000039339065552, |
| "sampling/importance_sampling_ratio/min": 0.28515318036079407, |
| "sampling/sampling_logp_difference/max": 1.2547287940979004, |
| "sampling/sampling_logp_difference/mean": 0.01729038543999195, |
| "step": 85 |
| }, |
| { |
| "clip_ratio/high_max": 0.00018562689729151316, |
| "clip_ratio/high_mean": 0.00018562689729151316, |
| "clip_ratio/low_mean": 8.496465306961909e-05, |
| "clip_ratio/low_min": 8.496465306961909e-05, |
| "clip_ratio/region_mean": 0.00027059155036113225, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14712.0, |
| "completions/max_terminated_length": 14712.0, |
| "completions/mean_length": 10643.125, |
| "completions/mean_terminated_length": 10643.125, |
| "completions/min_length": 7521.0, |
| "completions/min_terminated_length": 7521.0, |
| "entropy": 0.4584597982466221, |
| "epoch": 0.004943096907690539, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005199153907597065, |
| "learning_rate": 1e-05, |
| "loss": 0.1354, |
| "num_tokens": 4189781.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998418092727661, |
| "sampling/importance_sampling_ratio/min": 0.19236509501934052, |
| "sampling/sampling_logp_difference/max": 1.6483601331710815, |
| "sampling/sampling_logp_difference/mean": 0.017366668209433556, |
| "step": 86 |
| }, |
| { |
| "clip_ratio/high_max": 4.416961019160226e-05, |
| "clip_ratio/high_mean": 4.416961019160226e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.416961019160226e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2962.0, |
| "completions/max_terminated_length": 2962.0, |
| "completions/mean_length": 1943.625, |
| "completions/mean_terminated_length": 1943.625, |
| "completions/min_length": 1028.0, |
| "completions/min_terminated_length": 1028.0, |
| "entropy": 0.25489529222249985, |
| "epoch": 0.005000574778710197, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07604192942380905, |
| "learning_rate": 1e-05, |
| "loss": -0.0092, |
| "num_tokens": 4206050.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.397850751876831, |
| "sampling/importance_sampling_ratio/mean": 0.9999982118606567, |
| "sampling/importance_sampling_ratio/min": 0.6308754086494446, |
| "sampling/sampling_logp_difference/max": 0.4606468677520752, |
| "sampling/sampling_logp_difference/mean": 0.010172372683882713, |
| "step": 87 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013542795204557478, |
| "clip_ratio/high_mean": 0.00013542795204557478, |
| "clip_ratio/low_mean": 0.0006206220132298768, |
| "clip_ratio/low_min": 0.0006206220132298768, |
| "clip_ratio/region_mean": 0.0007560499652754515, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1202.0, |
| "completions/max_terminated_length": 1202.0, |
| "completions/mean_length": 763.5, |
| "completions/mean_terminated_length": 763.5, |
| "completions/min_length": 426.0, |
| "completions/min_terminated_length": 426.0, |
| "entropy": 0.33439876325428486, |
| "epoch": 0.005058052649729854, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.060591693967580795, |
| "learning_rate": 1e-05, |
| "loss": -0.0851, |
| "num_tokens": 4215230.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4514975547790527, |
| "sampling/importance_sampling_ratio/mean": 1.0007790327072144, |
| "sampling/importance_sampling_ratio/min": 0.6699815988540649, |
| "sampling/sampling_logp_difference/max": 0.40050506591796875, |
| "sampling/sampling_logp_difference/mean": 0.014445881359279156, |
| "step": 88 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1787.0, |
| "completions/max_terminated_length": 1787.0, |
| "completions/mean_length": 1103.875, |
| "completions/mean_terminated_length": 1103.875, |
| "completions/min_length": 481.0, |
| "completions/min_terminated_length": 481.0, |
| "entropy": 0.508196271955967, |
| "epoch": 0.005115530520749511, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 4225269.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.282193660736084, |
| "sampling/importance_sampling_ratio/mean": 1.000566005706787, |
| "sampling/importance_sampling_ratio/min": 0.6705912947654724, |
| "sampling/sampling_logp_difference/max": 0.3995954990386963, |
| "sampling/sampling_logp_difference/mean": 0.01484319195151329, |
| "step": 89 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.000270635835477151, |
| "clip_ratio/low_min": 0.000270635835477151, |
| "clip_ratio/region_mean": 0.000270635835477151, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 3458.0, |
| "completions/mean_length": 4003.5, |
| "completions/mean_terminated_length": 2234.857177734375, |
| "completions/min_length": 447.0, |
| "completions/min_terminated_length": 447.0, |
| "entropy": 0.4536377377808094, |
| "epoch": 0.005173008391769169, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.08329286426305771, |
| "learning_rate": 1e-05, |
| "loss": 0.4597, |
| "num_tokens": 4258313.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.8575853109359741, |
| "sampling/importance_sampling_ratio/mean": 1.0002970695495605, |
| "sampling/importance_sampling_ratio/min": 0.2592966556549072, |
| "sampling/sampling_logp_difference/max": 1.3497824668884277, |
| "sampling/sampling_logp_difference/mean": 0.020333070307970047, |
| "step": 90 |
| }, |
| { |
| "clip_ratio/high_max": 6.797360038035549e-05, |
| "clip_ratio/high_mean": 6.797360038035549e-05, |
| "clip_ratio/low_mean": 0.00015368337153631728, |
| "clip_ratio/low_min": 0.00015368337153631728, |
| "clip_ratio/region_mean": 0.00022165697191667277, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10391.0, |
| "completions/max_terminated_length": 10391.0, |
| "completions/mean_length": 5971.75, |
| "completions/mean_terminated_length": 5971.75, |
| "completions/min_length": 2969.0, |
| "completions/min_terminated_length": 2969.0, |
| "entropy": 0.4571814499795437, |
| "epoch": 0.005230486262788827, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022652029991149902, |
| "learning_rate": 1e-05, |
| "loss": 0.1659, |
| "num_tokens": 4306711.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6500544548034668, |
| "sampling/importance_sampling_ratio/mean": 1.0000139474868774, |
| "sampling/importance_sampling_ratio/min": 0.39687731862068176, |
| "sampling/sampling_logp_difference/max": 0.9241280555725098, |
| "sampling/sampling_logp_difference/mean": 0.018285445868968964, |
| "step": 91 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013248470713733695, |
| "clip_ratio/high_mean": 0.00013248470713733695, |
| "clip_ratio/low_mean": 0.00011607853230088949, |
| "clip_ratio/low_min": 0.00011607853230088949, |
| "clip_ratio/region_mean": 0.00024856323943822645, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7538.0, |
| "completions/max_terminated_length": 7538.0, |
| "completions/mean_length": 4436.125, |
| "completions/mean_terminated_length": 4436.125, |
| "completions/min_length": 2237.0, |
| "completions/min_terminated_length": 2237.0, |
| "entropy": 0.26149668358266354, |
| "epoch": 0.005287964133808483, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008685613982379436, |
| "learning_rate": 1e-05, |
| "loss": 0.2472, |
| "num_tokens": 4343240.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.5667108297348022, |
| "sampling/importance_sampling_ratio/mean": 1.0001031160354614, |
| "sampling/importance_sampling_ratio/min": 0.6066851615905762, |
| "sampling/sampling_logp_difference/max": 0.49974536895751953, |
| "sampling/sampling_logp_difference/mean": 0.010060268454253674, |
| "step": 92 |
| }, |
| { |
| "clip_ratio/high_max": 1.6438716556876898e-05, |
| "clip_ratio/high_mean": 1.6438716556876898e-05, |
| "clip_ratio/low_mean": 8.397716010222211e-05, |
| "clip_ratio/low_min": 8.397716010222211e-05, |
| "clip_ratio/region_mean": 0.00010041587665909901, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9654.0, |
| "completions/max_terminated_length": 9654.0, |
| "completions/mean_length": 4659.125, |
| "completions/mean_terminated_length": 4659.125, |
| "completions/min_length": 1628.0, |
| "completions/min_terminated_length": 1628.0, |
| "entropy": 0.43837379664182663, |
| "epoch": 0.005345442004828141, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.050454385578632355, |
| "learning_rate": 1e-05, |
| "loss": -0.5178, |
| "num_tokens": 4382209.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998786449432373, |
| "sampling/importance_sampling_ratio/min": 0.43582969903945923, |
| "sampling/sampling_logp_difference/max": 0.8305037021636963, |
| "sampling/sampling_logp_difference/mean": 0.014977103099226952, |
| "step": 93 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2592.0, |
| "completions/max_terminated_length": 2592.0, |
| "completions/mean_length": 1523.75, |
| "completions/mean_terminated_length": 1523.75, |
| "completions/min_length": 947.0, |
| "completions/min_terminated_length": 947.0, |
| "entropy": 0.263191694393754, |
| "epoch": 0.005402919875847799, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 4395311.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5041006803512573, |
| "sampling/importance_sampling_ratio/mean": 0.9999500513076782, |
| "sampling/importance_sampling_ratio/min": 0.6959434151649475, |
| "sampling/sampling_logp_difference/max": 0.4081951379776001, |
| "sampling/sampling_logp_difference/mean": 0.010398940183222294, |
| "step": 94 |
| }, |
| { |
| "clip_ratio/high_max": 7.324352191062644e-05, |
| "clip_ratio/high_mean": 7.324352191062644e-05, |
| "clip_ratio/low_mean": 0.00010513036249903962, |
| "clip_ratio/low_min": 0.00010513036249903962, |
| "clip_ratio/region_mean": 0.00017837388440966606, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3657.0, |
| "completions/max_terminated_length": 3657.0, |
| "completions/mean_length": 2862.875, |
| "completions/mean_terminated_length": 2862.875, |
| "completions/min_length": 1961.0, |
| "completions/min_terminated_length": 1961.0, |
| "entropy": 0.2586438972502947, |
| "epoch": 0.005460397746867456, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0077639296650886536, |
| "learning_rate": 1e-05, |
| "loss": -0.0599, |
| "num_tokens": 4419494.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.560463547706604, |
| "sampling/importance_sampling_ratio/mean": 1.0001232624053955, |
| "sampling/importance_sampling_ratio/min": 0.6484719514846802, |
| "sampling/sampling_logp_difference/max": 0.44498300552368164, |
| "sampling/sampling_logp_difference/mean": 0.010486197657883167, |
| "step": 95 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001295057481911499, |
| "clip_ratio/high_mean": 0.0001295057481911499, |
| "clip_ratio/low_mean": 0.0005240441350906622, |
| "clip_ratio/low_min": 0.0005240441350906622, |
| "clip_ratio/region_mean": 0.0006535498832818121, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11747.0, |
| "completions/max_terminated_length": 11747.0, |
| "completions/mean_length": 7692.625, |
| "completions/mean_terminated_length": 7692.625, |
| "completions/min_length": 4612.0, |
| "completions/min_terminated_length": 4612.0, |
| "entropy": 0.4370035044848919, |
| "epoch": 0.005517875617887113, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01865573227405548, |
| "learning_rate": 1e-05, |
| "loss": -0.0023, |
| "num_tokens": 4481763.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.9858016967773438, |
| "sampling/importance_sampling_ratio/mean": 0.9999425411224365, |
| "sampling/importance_sampling_ratio/min": 0.360934317111969, |
| "sampling/sampling_logp_difference/max": 1.0190593004226685, |
| "sampling/sampling_logp_difference/mean": 0.016827711835503578, |
| "step": 96 |
| }, |
| { |
| "clip_ratio/high_max": 1.5083865946508013e-05, |
| "clip_ratio/high_mean": 1.5083865946508013e-05, |
| "clip_ratio/low_mean": 0.00090244751481805, |
| "clip_ratio/low_min": 0.00090244751481805, |
| "clip_ratio/region_mean": 0.000917531380764558, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12921.0, |
| "completions/mean_length": 12423.625, |
| "completions/mean_terminated_length": 10047.400390625, |
| "completions/min_length": 7634.0, |
| "completions/min_terminated_length": 7634.0, |
| "entropy": 0.6236320100724697, |
| "epoch": 0.005575353488906771, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006804484408348799, |
| "learning_rate": 1e-05, |
| "loss": 0.118, |
| "num_tokens": 4582400.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997172355651855, |
| "sampling/importance_sampling_ratio/min": 0.3413894474506378, |
| "sampling/sampling_logp_difference/max": 1.0747313499450684, |
| "sampling/sampling_logp_difference/mean": 0.025754448026418686, |
| "step": 97 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15624.0, |
| "completions/mean_length": 13475.5, |
| "completions/mean_terminated_length": 12506.0, |
| "completions/min_length": 4824.0, |
| "completions/min_terminated_length": 4824.0, |
| "entropy": 0.4783651791512966, |
| "epoch": 0.0056328313599264285, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 4691060.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999643564224243, |
| "sampling/importance_sampling_ratio/min": 0.30113595724105835, |
| "sampling/sampling_logp_difference/max": 1.213273525238037, |
| "sampling/sampling_logp_difference/mean": 0.020278790965676308, |
| "step": 98 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.6158835680689663e-05, |
| "clip_ratio/low_min": 2.6158835680689663e-05, |
| "clip_ratio/region_mean": 2.6158835680689663e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9557.0, |
| "completions/max_terminated_length": 9557.0, |
| "completions/mean_length": 2165.875, |
| "completions/mean_terminated_length": 2165.875, |
| "completions/min_length": 794.0, |
| "completions/min_terminated_length": 794.0, |
| "entropy": 0.29633556492626667, |
| "epoch": 0.005690309230946085, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013613307848572731, |
| "learning_rate": 1e-05, |
| "loss": 0.784, |
| "num_tokens": 4709435.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.7876187562942505, |
| "sampling/importance_sampling_ratio/mean": 0.999664306640625, |
| "sampling/importance_sampling_ratio/min": 0.6190581917762756, |
| "sampling/sampling_logp_difference/max": 0.5808844566345215, |
| "sampling/sampling_logp_difference/mean": 0.013285611756145954, |
| "step": 99 |
| }, |
| { |
| "clip_ratio/high_max": 9.427890472579747e-05, |
| "clip_ratio/high_mean": 9.427890472579747e-05, |
| "clip_ratio/low_mean": 0.0008690841023053508, |
| "clip_ratio/low_min": 0.0008690841023053508, |
| "clip_ratio/region_mean": 0.0009633630070311483, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14278.0, |
| "completions/mean_length": 12535.125, |
| "completions/mean_terminated_length": 11985.2861328125, |
| "completions/min_length": 6118.0, |
| "completions/min_terminated_length": 6118.0, |
| "entropy": 0.5694211684167385, |
| "epoch": 0.005747787101965743, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006562459748238325, |
| "learning_rate": 1e-05, |
| "loss": 0.0231, |
| "num_tokens": 4810828.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003129243850708, |
| "sampling/importance_sampling_ratio/min": 0.2958182096481323, |
| "sampling/sampling_logp_difference/max": 1.2180101871490479, |
| "sampling/sampling_logp_difference/mean": 0.025095155462622643, |
| "step": 100 |
| }, |
| { |
| "clip_ratio/high_max": 0.00021666069733328186, |
| "clip_ratio/high_mean": 0.00021666069733328186, |
| "clip_ratio/low_mean": 0.00034195535408798605, |
| "clip_ratio/low_min": 0.00034195535408798605, |
| "clip_ratio/region_mean": 0.0005586160514212679, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5100.0, |
| "completions/max_terminated_length": 5100.0, |
| "completions/mean_length": 3674.875, |
| "completions/mean_terminated_length": 3674.875, |
| "completions/min_length": 609.0, |
| "completions/min_terminated_length": 609.0, |
| "entropy": 0.6064490713179111, |
| "epoch": 0.005805264972985401, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.04777848348021507, |
| "learning_rate": 1e-05, |
| "loss": 0.0129, |
| "num_tokens": 4841243.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4659724235534668, |
| "sampling/importance_sampling_ratio/mean": 1.0000799894332886, |
| "sampling/importance_sampling_ratio/min": 0.6710418462753296, |
| "sampling/sampling_logp_difference/max": 0.39892375469207764, |
| "sampling/sampling_logp_difference/mean": 0.023423591628670692, |
| "step": 101 |
| }, |
| { |
| "clip_ratio/high_max": 0.00020406878411449725, |
| "clip_ratio/high_mean": 0.00020406878411449725, |
| "clip_ratio/low_mean": 0.00021950393420411274, |
| "clip_ratio/low_min": 0.00021950393420411274, |
| "clip_ratio/region_mean": 0.00042357271831861, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9886.0, |
| "completions/max_terminated_length": 9886.0, |
| "completions/mean_length": 5910.5, |
| "completions/mean_terminated_length": 5910.5, |
| "completions/min_length": 842.0, |
| "completions/min_terminated_length": 842.0, |
| "entropy": 0.5742521323263645, |
| "epoch": 0.005862742844005058, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02844126522541046, |
| "learning_rate": 1e-05, |
| "loss": -0.2702, |
| "num_tokens": 4889535.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6540199518203735, |
| "sampling/importance_sampling_ratio/mean": 0.9999129176139832, |
| "sampling/importance_sampling_ratio/min": 0.4713459610939026, |
| "sampling/sampling_logp_difference/max": 0.7521629333496094, |
| "sampling/sampling_logp_difference/mean": 0.023292189463973045, |
| "step": 102 |
| }, |
| { |
| "clip_ratio/high_max": 7.457700303348247e-05, |
| "clip_ratio/high_mean": 7.457700303348247e-05, |
| "clip_ratio/low_mean": 0.0003259427976445295, |
| "clip_ratio/low_min": 0.0003259427976445295, |
| "clip_ratio/region_mean": 0.00040051980067801196, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14820.0, |
| "completions/mean_length": 11594.75, |
| "completions/mean_terminated_length": 10910.572265625, |
| "completions/min_length": 6098.0, |
| "completions/min_terminated_length": 6098.0, |
| "entropy": 0.3187015615403652, |
| "epoch": 0.005920220715024715, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014056019484996796, |
| "learning_rate": 1e-05, |
| "loss": 0.1221, |
| "num_tokens": 4983725.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999797344207764, |
| "sampling/importance_sampling_ratio/min": 0.0509355291724205, |
| "sampling/sampling_logp_difference/max": 2.9771945476531982, |
| "sampling/sampling_logp_difference/mean": 0.015568562783300877, |
| "step": 103 |
| }, |
| { |
| "clip_ratio/high_max": 7.59027407184476e-05, |
| "clip_ratio/high_mean": 7.59027407184476e-05, |
| "clip_ratio/low_mean": 0.00043579248449532315, |
| "clip_ratio/low_min": 0.00043579248449532315, |
| "clip_ratio/region_mean": 0.0005116952252137708, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10378.0, |
| "completions/max_terminated_length": 10378.0, |
| "completions/mean_length": 6279.375, |
| "completions/mean_terminated_length": 6279.375, |
| "completions/min_length": 3401.0, |
| "completions/min_terminated_length": 3401.0, |
| "entropy": 0.6004678979516029, |
| "epoch": 0.005977698586044373, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.037188686430454254, |
| "learning_rate": 1e-05, |
| "loss": 0.1846, |
| "num_tokens": 5035672.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001723766326904, |
| "sampling/importance_sampling_ratio/min": 0.39308810234069824, |
| "sampling/sampling_logp_difference/max": 1.280113697052002, |
| "sampling/sampling_logp_difference/mean": 0.02408575639128685, |
| "step": 104 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00020284376842027996, |
| "clip_ratio/low_min": 0.00020284376842027996, |
| "clip_ratio/region_mean": 0.00020284376842027996, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12887.0, |
| "completions/max_terminated_length": 12887.0, |
| "completions/mean_length": 8273.75, |
| "completions/mean_terminated_length": 8273.75, |
| "completions/min_length": 3123.0, |
| "completions/min_terminated_length": 3123.0, |
| "entropy": 0.8106596171855927, |
| "epoch": 0.0060351764570640305, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019368216395378113, |
| "learning_rate": 1e-05, |
| "loss": 0.3, |
| "num_tokens": 5102726.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003316402435303, |
| "sampling/importance_sampling_ratio/min": 0.37379640340805054, |
| "sampling/sampling_logp_difference/max": 0.984044075012207, |
| "sampling/sampling_logp_difference/mean": 0.02187422849237919, |
| "step": 105 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4021.0, |
| "completions/max_terminated_length": 4021.0, |
| "completions/mean_length": 2854.375, |
| "completions/mean_terminated_length": 2854.375, |
| "completions/min_length": 2024.0, |
| "completions/min_terminated_length": 2024.0, |
| "entropy": 0.2449374794960022, |
| "epoch": 0.006092654328083688, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008184988051652908, |
| "learning_rate": 1e-05, |
| "loss": -0.086, |
| "num_tokens": 5126889.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3411176204681396, |
| "sampling/importance_sampling_ratio/mean": 0.9995095729827881, |
| "sampling/importance_sampling_ratio/min": 0.7121059894561768, |
| "sampling/sampling_logp_difference/max": 0.33952856063842773, |
| "sampling/sampling_logp_difference/mean": 0.010186538100242615, |
| "step": 106 |
| }, |
| { |
| "clip_ratio/high_max": 0.00023311110089707654, |
| "clip_ratio/high_mean": 0.00023311110089707654, |
| "clip_ratio/low_mean": 0.0005337549955584109, |
| "clip_ratio/low_min": 0.0005337549955584109, |
| "clip_ratio/region_mean": 0.0007668660964554874, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7455.0, |
| "completions/max_terminated_length": 7455.0, |
| "completions/mean_length": 5222.0, |
| "completions/mean_terminated_length": 5222.0, |
| "completions/min_length": 2643.0, |
| "completions/min_terminated_length": 2643.0, |
| "entropy": 0.7336124926805496, |
| "epoch": 0.006150132199103345, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.036571402102708817, |
| "learning_rate": 1e-05, |
| "loss": 0.0096, |
| "num_tokens": 5170073.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6328125, |
| "sampling/importance_sampling_ratio/mean": 0.9998865127563477, |
| "sampling/importance_sampling_ratio/min": 0.4239436388015747, |
| "sampling/sampling_logp_difference/max": 0.8581547737121582, |
| "sampling/sampling_logp_difference/mean": 0.027363304048776627, |
| "step": 107 |
| }, |
| { |
| "clip_ratio/high_max": 5.095334017823916e-05, |
| "clip_ratio/high_mean": 5.095334017823916e-05, |
| "clip_ratio/low_mean": 0.00020405950272106566, |
| "clip_ratio/low_min": 0.00020405950272106566, |
| "clip_ratio/region_mean": 0.0002550128428993048, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7622.0, |
| "completions/max_terminated_length": 7622.0, |
| "completions/mean_length": 5145.75, |
| "completions/mean_terminated_length": 5145.75, |
| "completions/min_length": 3266.0, |
| "completions/min_terminated_length": 3266.0, |
| "entropy": 0.4340439885854721, |
| "epoch": 0.006207610070123003, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.04015190154314041, |
| "learning_rate": 1e-05, |
| "loss": 0.0621, |
| "num_tokens": 5212927.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.7912691831588745, |
| "sampling/importance_sampling_ratio/mean": 0.999870240688324, |
| "sampling/importance_sampling_ratio/min": 0.33116602897644043, |
| "sampling/sampling_logp_difference/max": 1.105135440826416, |
| "sampling/sampling_logp_difference/mean": 0.014029542915523052, |
| "step": 108 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00010089762690768111, |
| "clip_ratio/low_min": 0.00010089762690768111, |
| "clip_ratio/region_mean": 0.00010089762690768111, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9632.0, |
| "completions/max_terminated_length": 9632.0, |
| "completions/mean_length": 5762.5, |
| "completions/mean_terminated_length": 5762.5, |
| "completions/min_length": 3287.0, |
| "completions/min_terminated_length": 3287.0, |
| "entropy": 0.39456985145807266, |
| "epoch": 0.00626508794114266, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009497498162090778, |
| "learning_rate": 1e-05, |
| "loss": 0.1078, |
| "num_tokens": 5260299.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001640319824219, |
| "sampling/importance_sampling_ratio/min": 0.3906859755516052, |
| "sampling/sampling_logp_difference/max": 1.2457599639892578, |
| "sampling/sampling_logp_difference/mean": 0.013365420512855053, |
| "step": 109 |
| }, |
| { |
| "clip_ratio/high_max": 6.160670454846695e-05, |
| "clip_ratio/high_mean": 6.160670454846695e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 6.160670454846695e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2574.0, |
| "completions/max_terminated_length": 2574.0, |
| "completions/mean_length": 1961.125, |
| "completions/mean_terminated_length": 1961.125, |
| "completions/min_length": 1186.0, |
| "completions/min_terminated_length": 1186.0, |
| "entropy": 0.20434438437223434, |
| "epoch": 0.006322565812162318, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.054287102073431015, |
| "learning_rate": 1e-05, |
| "loss": -0.1395, |
| "num_tokens": 5277180.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.9964170455932617, |
| "sampling/importance_sampling_ratio/mean": 0.9997534155845642, |
| "sampling/importance_sampling_ratio/min": 0.6984014511108398, |
| "sampling/sampling_logp_difference/max": 0.6913540363311768, |
| "sampling/sampling_logp_difference/mean": 0.008898675441741943, |
| "step": 110 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012034870269417297, |
| "clip_ratio/high_mean": 0.00012034870269417297, |
| "clip_ratio/low_mean": 5.5915901612024754e-05, |
| "clip_ratio/low_min": 5.5915901612024754e-05, |
| "clip_ratio/region_mean": 0.00017626460430619773, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9815.0, |
| "completions/max_terminated_length": 9815.0, |
| "completions/mean_length": 7040.375, |
| "completions/mean_terminated_length": 7040.375, |
| "completions/min_length": 4229.0, |
| "completions/min_terminated_length": 4229.0, |
| "entropy": 0.2732951082289219, |
| "epoch": 0.006380043683181975, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006093955133110285, |
| "learning_rate": 1e-05, |
| "loss": -0.1292, |
| "num_tokens": 5334319.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.8080084323883057, |
| "sampling/importance_sampling_ratio/mean": 1.0001444816589355, |
| "sampling/importance_sampling_ratio/min": 0.40781381726264954, |
| "sampling/sampling_logp_difference/max": 0.896944522857666, |
| "sampling/sampling_logp_difference/mean": 0.011279252357780933, |
| "step": 111 |
| }, |
| { |
| "clip_ratio/high_max": 3.80923374905251e-05, |
| "clip_ratio/high_mean": 3.80923374905251e-05, |
| "clip_ratio/low_mean": 0.0008192299137590453, |
| "clip_ratio/low_min": 0.0008192299137590453, |
| "clip_ratio/region_mean": 0.0008573222512495704, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7425.0, |
| "completions/max_terminated_length": 7425.0, |
| "completions/mean_length": 5289.25, |
| "completions/mean_terminated_length": 5289.25, |
| "completions/min_length": 3101.0, |
| "completions/min_terminated_length": 3101.0, |
| "entropy": 0.5866907760500908, |
| "epoch": 0.006437521554201632, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02391251176595688, |
| "learning_rate": 1e-05, |
| "loss": -0.0641, |
| "num_tokens": 5377417.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.7714844942092896, |
| "sampling/importance_sampling_ratio/mean": 1.0000824928283691, |
| "sampling/importance_sampling_ratio/min": 0.5712507367134094, |
| "sampling/sampling_logp_difference/max": 0.5718178749084473, |
| "sampling/sampling_logp_difference/mean": 0.023795749992132187, |
| "step": 112 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7667.0, |
| "completions/max_terminated_length": 7667.0, |
| "completions/mean_length": 4092.25, |
| "completions/mean_terminated_length": 4092.25, |
| "completions/min_length": 2210.0, |
| "completions/min_terminated_length": 2210.0, |
| "entropy": 0.26948732510209084, |
| "epoch": 0.00649499942522129, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 5411419.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5129711627960205, |
| "sampling/importance_sampling_ratio/mean": 1.0002878904342651, |
| "sampling/importance_sampling_ratio/min": 0.6073850393295288, |
| "sampling/sampling_logp_difference/max": 0.4985923767089844, |
| "sampling/sampling_logp_difference/mean": 0.011067938059568405, |
| "step": 113 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10053.0, |
| "completions/max_terminated_length": 10053.0, |
| "completions/mean_length": 6838.0, |
| "completions/mean_terminated_length": 6838.0, |
| "completions/min_length": 4493.0, |
| "completions/min_terminated_length": 4493.0, |
| "entropy": 0.36648314259946346, |
| "epoch": 0.006552477296240947, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 5467379.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999701976776123, |
| "sampling/importance_sampling_ratio/min": 0.34160375595092773, |
| "sampling/sampling_logp_difference/max": 1.074103832244873, |
| "sampling/sampling_logp_difference/mean": 0.01511444803327322, |
| "step": 114 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002783380914479494, |
| "clip_ratio/high_mean": 0.0002783380914479494, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0002783380914479494, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2412.0, |
| "completions/max_terminated_length": 2412.0, |
| "completions/mean_length": 1630.5, |
| "completions/mean_terminated_length": 1630.5, |
| "completions/min_length": 910.0, |
| "completions/min_terminated_length": 910.0, |
| "entropy": 0.3034412022680044, |
| "epoch": 0.0066099551672606045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01175200566649437, |
| "learning_rate": 1e-05, |
| "loss": -0.0583, |
| "num_tokens": 5481591.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3706064224243164, |
| "sampling/importance_sampling_ratio/mean": 0.9998339414596558, |
| "sampling/importance_sampling_ratio/min": 0.2978171408176422, |
| "sampling/sampling_logp_difference/max": 1.211275577545166, |
| "sampling/sampling_logp_difference/mean": 0.011887015774846077, |
| "step": 115 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00022903785429662094, |
| "clip_ratio/low_min": 0.00022903785429662094, |
| "clip_ratio/region_mean": 0.00022903785429662094, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1171.0, |
| "completions/max_terminated_length": 1171.0, |
| "completions/mean_length": 1044.375, |
| "completions/mean_terminated_length": 1044.375, |
| "completions/min_length": 936.0, |
| "completions/min_terminated_length": 936.0, |
| "entropy": 0.3192705698311329, |
| "epoch": 0.006667433038280262, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05706014484167099, |
| "learning_rate": 1e-05, |
| "loss": -0.0136, |
| "num_tokens": 5491114.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.4043923616409302, |
| "sampling/importance_sampling_ratio/mean": 1.0009198188781738, |
| "sampling/importance_sampling_ratio/min": 0.6967563033103943, |
| "sampling/sampling_logp_difference/max": 0.36131954193115234, |
| "sampling/sampling_logp_difference/mean": 0.012880105525255203, |
| "step": 116 |
| }, |
| { |
| "clip_ratio/high_max": 2.0398172637214884e-05, |
| "clip_ratio/high_mean": 2.0398172637214884e-05, |
| "clip_ratio/low_mean": 0.000778431014623493, |
| "clip_ratio/low_min": 0.000778431014623493, |
| "clip_ratio/region_mean": 0.0007988291872607078, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15170.0, |
| "completions/mean_length": 9836.875, |
| "completions/mean_terminated_length": 8901.572265625, |
| "completions/min_length": 5382.0, |
| "completions/min_terminated_length": 5382.0, |
| "entropy": 0.5065955854952335, |
| "epoch": 0.00672491090929992, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0058839512057602406, |
| "learning_rate": 1e-05, |
| "loss": 0.1335, |
| "num_tokens": 5570745.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003448724746704, |
| "sampling/importance_sampling_ratio/min": 0.30787843465805054, |
| "sampling/sampling_logp_difference/max": 2.0774405002593994, |
| "sampling/sampling_logp_difference/mean": 0.0221853144466877, |
| "step": 117 |
| }, |
| { |
| "clip_ratio/high_max": 6.92520770826377e-05, |
| "clip_ratio/high_mean": 6.92520770826377e-05, |
| "clip_ratio/low_mean": 6.154603761387989e-05, |
| "clip_ratio/low_min": 6.154603761387989e-05, |
| "clip_ratio/region_mean": 0.0001307981146965176, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2781.0, |
| "completions/max_terminated_length": 2781.0, |
| "completions/mean_length": 1831.625, |
| "completions/mean_terminated_length": 1831.625, |
| "completions/min_length": 1006.0, |
| "completions/min_terminated_length": 1006.0, |
| "entropy": 0.57236672565341, |
| "epoch": 0.006782388780319577, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.015471331775188446, |
| "learning_rate": 1e-05, |
| "loss": 0.0389, |
| "num_tokens": 5586390.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.295737385749817, |
| "sampling/importance_sampling_ratio/mean": 1.0000808238983154, |
| "sampling/importance_sampling_ratio/min": 0.7419808506965637, |
| "sampling/sampling_logp_difference/max": 0.2984318733215332, |
| "sampling/sampling_logp_difference/mean": 0.015503806062042713, |
| "step": 118 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015082918798725586, |
| "clip_ratio/high_mean": 0.00015082918798725586, |
| "clip_ratio/low_mean": 0.00022748402261640877, |
| "clip_ratio/low_min": 0.00022748402261640877, |
| "clip_ratio/region_mean": 0.00037831321060366463, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16313.0, |
| "completions/mean_length": 9942.75, |
| "completions/mean_terminated_length": 9022.572265625, |
| "completions/min_length": 2978.0, |
| "completions/min_terminated_length": 2978.0, |
| "entropy": 0.474183764308691, |
| "epoch": 0.006839866651339234, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004840504843741655, |
| "learning_rate": 1e-05, |
| "loss": 0.2649, |
| "num_tokens": 5666788.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997504949569702, |
| "sampling/importance_sampling_ratio/min": 0.3404845893383026, |
| "sampling/sampling_logp_difference/max": 1.077385425567627, |
| "sampling/sampling_logp_difference/mean": 0.020352305844426155, |
| "step": 119 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00025774889945751056, |
| "clip_ratio/low_min": 0.00025774889945751056, |
| "clip_ratio/region_mean": 0.00025774889945751056, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1997.0, |
| "completions/max_terminated_length": 1997.0, |
| "completions/mean_length": 1550.375, |
| "completions/mean_terminated_length": 1550.375, |
| "completions/min_length": 978.0, |
| "completions/min_terminated_length": 978.0, |
| "entropy": 0.33403414487838745, |
| "epoch": 0.006897344522358892, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.04866841435432434, |
| "learning_rate": 1e-05, |
| "loss": 0.0581, |
| "num_tokens": 5680391.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.364777684211731, |
| "sampling/importance_sampling_ratio/mean": 0.9999312162399292, |
| "sampling/importance_sampling_ratio/min": 0.6394995450973511, |
| "sampling/sampling_logp_difference/max": 0.4470694065093994, |
| "sampling/sampling_logp_difference/mean": 0.013162706978619099, |
| "step": 120 |
| }, |
| { |
| "clip_ratio/high_max": 6.664296961389482e-05, |
| "clip_ratio/high_mean": 6.664296961389482e-05, |
| "clip_ratio/low_mean": 0.0003867654777423013, |
| "clip_ratio/low_min": 0.0003867654777423013, |
| "clip_ratio/region_mean": 0.0004534084473561961, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15640.0, |
| "completions/max_terminated_length": 15640.0, |
| "completions/mean_length": 7863.375, |
| "completions/mean_terminated_length": 7863.375, |
| "completions/min_length": 2762.0, |
| "completions/min_terminated_length": 2762.0, |
| "entropy": 0.43769025802612305, |
| "epoch": 0.00695482239337855, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03385671600699425, |
| "learning_rate": 1e-05, |
| "loss": 0.1263, |
| "num_tokens": 5744842.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997232556343079, |
| "sampling/importance_sampling_ratio/min": 0.1058628186583519, |
| "sampling/sampling_logp_difference/max": 2.2456111907958984, |
| "sampling/sampling_logp_difference/mean": 0.020388441160321236, |
| "step": 121 |
| }, |
| { |
| "clip_ratio/high_max": 8.19703327579191e-05, |
| "clip_ratio/high_mean": 8.19703327579191e-05, |
| "clip_ratio/low_mean": 0.00010615710925776511, |
| "clip_ratio/low_min": 0.00010615710925776511, |
| "clip_ratio/region_mean": 0.0001881274420156842, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8899.0, |
| "completions/max_terminated_length": 8899.0, |
| "completions/mean_length": 5869.375, |
| "completions/mean_terminated_length": 5869.375, |
| "completions/min_length": 3585.0, |
| "completions/min_terminated_length": 3585.0, |
| "entropy": 0.3783104941248894, |
| "epoch": 0.007012300264398206, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0070150443352758884, |
| "learning_rate": 1e-05, |
| "loss": -0.0695, |
| "num_tokens": 5793069.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4635522365570068, |
| "sampling/importance_sampling_ratio/mean": 1.0000921487808228, |
| "sampling/importance_sampling_ratio/min": 0.5364006161689758, |
| "sampling/sampling_logp_difference/max": 0.6228740215301514, |
| "sampling/sampling_logp_difference/mean": 0.014711553230881691, |
| "step": 122 |
| }, |
| { |
| "clip_ratio/high_max": 1.9290124328108504e-05, |
| "clip_ratio/high_mean": 1.9290124328108504e-05, |
| "clip_ratio/low_mean": 0.0003349564867676236, |
| "clip_ratio/low_min": 0.0003349564867676236, |
| "clip_ratio/region_mean": 0.0003542466110957321, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9364.0, |
| "completions/max_terminated_length": 9364.0, |
| "completions/mean_length": 4458.625, |
| "completions/mean_terminated_length": 4458.625, |
| "completions/min_length": 1577.0, |
| "completions/min_terminated_length": 1577.0, |
| "entropy": 0.6041946038603783, |
| "epoch": 0.007069778135417864, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.045138485729694366, |
| "learning_rate": 1e-05, |
| "loss": 0.1476, |
| "num_tokens": 5830658.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4898751974105835, |
| "sampling/importance_sampling_ratio/mean": 1.0004459619522095, |
| "sampling/importance_sampling_ratio/min": 0.5976069569587708, |
| "sampling/sampling_logp_difference/max": 0.5148220062255859, |
| "sampling/sampling_logp_difference/mean": 0.019251661375164986, |
| "step": 123 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0002357012745051179, |
| "clip_ratio/low_min": 0.0002357012745051179, |
| "clip_ratio/region_mean": 0.0002357012745051179, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3192.0, |
| "completions/max_terminated_length": 3192.0, |
| "completions/mean_length": 1549.375, |
| "completions/mean_terminated_length": 1549.375, |
| "completions/min_length": 946.0, |
| "completions/min_terminated_length": 946.0, |
| "entropy": 0.3446086347103119, |
| "epoch": 0.007127256006437522, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.025366822257637978, |
| "learning_rate": 1e-05, |
| "loss": 0.3489, |
| "num_tokens": 5843965.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.5565142631530762, |
| "sampling/importance_sampling_ratio/mean": 0.9995758533477783, |
| "sampling/importance_sampling_ratio/min": 0.6307206749916077, |
| "sampling/sampling_logp_difference/max": 0.4608922004699707, |
| "sampling/sampling_logp_difference/mean": 0.015049204230308533, |
| "step": 124 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4179.0, |
| "completions/max_terminated_length": 4179.0, |
| "completions/mean_length": 3381.5, |
| "completions/mean_terminated_length": 3381.5, |
| "completions/min_length": 2415.0, |
| "completions/min_terminated_length": 2415.0, |
| "entropy": 0.23659361898899078, |
| "epoch": 0.007184733877457179, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 5872265.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3590935468673706, |
| "sampling/importance_sampling_ratio/mean": 1.0000966787338257, |
| "sampling/importance_sampling_ratio/min": 0.6720110774040222, |
| "sampling/sampling_logp_difference/max": 0.39748048782348633, |
| "sampling/sampling_logp_difference/mean": 0.009295837953686714, |
| "step": 125 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1474.0, |
| "completions/max_terminated_length": 1474.0, |
| "completions/mean_length": 1049.625, |
| "completions/mean_terminated_length": 1049.625, |
| "completions/min_length": 845.0, |
| "completions/min_terminated_length": 845.0, |
| "entropy": 0.30449257232248783, |
| "epoch": 0.007242211748476836, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 5881782.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.395222783088684, |
| "sampling/importance_sampling_ratio/mean": 1.0002799034118652, |
| "sampling/importance_sampling_ratio/min": 0.6292243599891663, |
| "sampling/sampling_logp_difference/max": 0.46326732635498047, |
| "sampling/sampling_logp_difference/mean": 0.01346125639975071, |
| "step": 126 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0004151017274125479, |
| "clip_ratio/low_min": 0.0004151017274125479, |
| "clip_ratio/region_mean": 0.0004151017274125479, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9615.0, |
| "completions/max_terminated_length": 9615.0, |
| "completions/mean_length": 2985.625, |
| "completions/mean_terminated_length": 2985.625, |
| "completions/min_length": 1412.0, |
| "completions/min_terminated_length": 1412.0, |
| "entropy": 0.5704441592097282, |
| "epoch": 0.007299689619496494, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.020893597975373268, |
| "learning_rate": 1e-05, |
| "loss": 0.175, |
| "num_tokens": 5906651.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.8009577989578247, |
| "sampling/importance_sampling_ratio/mean": 1.0002970695495605, |
| "sampling/importance_sampling_ratio/min": 0.5241647958755493, |
| "sampling/sampling_logp_difference/max": 0.645949125289917, |
| "sampling/sampling_logp_difference/mean": 0.023233845829963684, |
| "step": 127 |
| }, |
| { |
| "clip_ratio/high_max": 0.00020371542268549092, |
| "clip_ratio/high_mean": 0.00020371542268549092, |
| "clip_ratio/low_mean": 0.000363931103493087, |
| "clip_ratio/low_min": 0.000363931103493087, |
| "clip_ratio/region_mean": 0.0005676465261785779, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14664.0, |
| "completions/max_terminated_length": 14664.0, |
| "completions/mean_length": 11802.625, |
| "completions/mean_terminated_length": 11802.625, |
| "completions/min_length": 7176.0, |
| "completions/min_terminated_length": 7176.0, |
| "entropy": 0.6763871014118195, |
| "epoch": 0.0073571674905161515, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009250272996723652, |
| "learning_rate": 1e-05, |
| "loss": 0.1183, |
| "num_tokens": 6002128.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000865459442139, |
| "sampling/importance_sampling_ratio/min": 0.2995777130126953, |
| "sampling/sampling_logp_difference/max": 1.2053813934326172, |
| "sampling/sampling_logp_difference/mean": 0.02760196290910244, |
| "step": 128 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012694765064225066, |
| "clip_ratio/high_mean": 0.00012694765064225066, |
| "clip_ratio/low_mean": 0.00015172972052823752, |
| "clip_ratio/low_min": 0.00015172972052823752, |
| "clip_ratio/region_mean": 0.0002786773711704882, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10329.0, |
| "completions/max_terminated_length": 10329.0, |
| "completions/mean_length": 5570.75, |
| "completions/mean_terminated_length": 5570.75, |
| "completions/min_length": 3491.0, |
| "completions/min_terminated_length": 3491.0, |
| "entropy": 0.45906075090169907, |
| "epoch": 0.007414645361535808, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008599936030805111, |
| "learning_rate": 1e-05, |
| "loss": -0.0399, |
| "num_tokens": 6047822.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.8152507543563843, |
| "sampling/importance_sampling_ratio/mean": 0.9997912049293518, |
| "sampling/importance_sampling_ratio/min": 0.4613805413246155, |
| "sampling/sampling_logp_difference/max": 0.7735320925712585, |
| "sampling/sampling_logp_difference/mean": 0.017998823896050453, |
| "step": 129 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.75, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14605.0, |
| "completions/mean_length": 15545.875, |
| "completions/mean_terminated_length": 13031.5, |
| "completions/min_length": 11458.0, |
| "completions/min_terminated_length": 11458.0, |
| "entropy": 0.19472824409604073, |
| "epoch": 0.007472123232555466, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6173981.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001029968261719, |
| "sampling/importance_sampling_ratio/min": 0.310546338558197, |
| "sampling/sampling_logp_difference/max": 1.1694221496582031, |
| "sampling/sampling_logp_difference/mean": 0.00943661853671074, |
| "step": 130 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9302.0, |
| "completions/max_terminated_length": 9302.0, |
| "completions/mean_length": 5712.5, |
| "completions/mean_terminated_length": 5712.5, |
| "completions/min_length": 3497.0, |
| "completions/min_terminated_length": 3497.0, |
| "entropy": 0.6666118577122688, |
| "epoch": 0.007529601103575124, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6220553.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000160932540894, |
| "sampling/importance_sampling_ratio/min": 0.27957940101623535, |
| "sampling/sampling_logp_difference/max": 1.2744688987731934, |
| "sampling/sampling_logp_difference/mean": 0.025103742256760597, |
| "step": 131 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2163.0, |
| "completions/max_terminated_length": 2163.0, |
| "completions/mean_length": 1661.625, |
| "completions/mean_terminated_length": 1661.625, |
| "completions/min_length": 1090.0, |
| "completions/min_terminated_length": 1090.0, |
| "entropy": 0.15260465070605278, |
| "epoch": 0.007587078974594781, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6234718.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.404661774635315, |
| "sampling/importance_sampling_ratio/mean": 1.0002719163894653, |
| "sampling/importance_sampling_ratio/min": 0.6761270761489868, |
| "sampling/sampling_logp_difference/max": 0.39137423038482666, |
| "sampling/sampling_logp_difference/mean": 0.006605846807360649, |
| "step": 132 |
| }, |
| { |
| "clip_ratio/high_max": 5.024858546676114e-05, |
| "clip_ratio/high_mean": 5.024858546676114e-05, |
| "clip_ratio/low_mean": 0.0008835435146465898, |
| "clip_ratio/low_min": 0.0008835435146465898, |
| "clip_ratio/region_mean": 0.0009337921001133509, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13484.0, |
| "completions/mean_length": 7938.125, |
| "completions/mean_terminated_length": 6731.57177734375, |
| "completions/min_length": 3555.0, |
| "completions/min_terminated_length": 3555.0, |
| "entropy": 0.3728845566511154, |
| "epoch": 0.007644556845614438, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007273501250892878, |
| "learning_rate": 1e-05, |
| "loss": 0.1902, |
| "num_tokens": 6299279.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000244140625, |
| "sampling/importance_sampling_ratio/min": 0.44445252418518066, |
| "sampling/sampling_logp_difference/max": 0.9838747978210449, |
| "sampling/sampling_logp_difference/mean": 0.016618642956018448, |
| "step": 133 |
| }, |
| { |
| "clip_ratio/high_max": 6.319514795904979e-05, |
| "clip_ratio/high_mean": 6.319514795904979e-05, |
| "clip_ratio/low_mean": 5.792400406789966e-05, |
| "clip_ratio/low_min": 5.792400406789966e-05, |
| "clip_ratio/region_mean": 0.00012111915202694945, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2867.0, |
| "completions/max_terminated_length": 2867.0, |
| "completions/mean_length": 2057.25, |
| "completions/mean_terminated_length": 2057.25, |
| "completions/min_length": 1652.0, |
| "completions/min_terminated_length": 1652.0, |
| "entropy": 0.28987678699195385, |
| "epoch": 0.007702034716634096, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01631118729710579, |
| "learning_rate": 1e-05, |
| "loss": -0.0321, |
| "num_tokens": 6316705.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4888286590576172, |
| "sampling/importance_sampling_ratio/mean": 1.000054955482483, |
| "sampling/importance_sampling_ratio/min": 0.6797762513160706, |
| "sampling/sampling_logp_difference/max": 0.39798974990844727, |
| "sampling/sampling_logp_difference/mean": 0.011149604804813862, |
| "step": 134 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7552.0, |
| "completions/max_terminated_length": 7552.0, |
| "completions/mean_length": 3074.625, |
| "completions/mean_terminated_length": 3074.625, |
| "completions/min_length": 1472.0, |
| "completions/min_terminated_length": 1472.0, |
| "entropy": 0.19418606348335743, |
| "epoch": 0.0077595125876537534, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6342766.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5750576257705688, |
| "sampling/importance_sampling_ratio/mean": 1.000051498413086, |
| "sampling/importance_sampling_ratio/min": 0.6158427596092224, |
| "sampling/sampling_logp_difference/max": 0.48476362228393555, |
| "sampling/sampling_logp_difference/mean": 0.00793286319822073, |
| "step": 135 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4646.0, |
| "completions/max_terminated_length": 4646.0, |
| "completions/mean_length": 3114.375, |
| "completions/mean_terminated_length": 3114.375, |
| "completions/min_length": 1475.0, |
| "completions/min_terminated_length": 1475.0, |
| "entropy": 0.5646222084760666, |
| "epoch": 0.007816990458673411, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6368873.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4438854455947876, |
| "sampling/importance_sampling_ratio/mean": 0.9999485611915588, |
| "sampling/importance_sampling_ratio/min": 0.6187267303466797, |
| "sampling/sampling_logp_difference/max": 0.48009157180786133, |
| "sampling/sampling_logp_difference/mean": 0.020424988120794296, |
| "step": 136 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0003590989581425674, |
| "clip_ratio/low_min": 0.0003590989581425674, |
| "clip_ratio/region_mean": 0.0003590989581425674, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1278.0, |
| "completions/max_terminated_length": 1278.0, |
| "completions/mean_length": 970.625, |
| "completions/mean_terminated_length": 970.625, |
| "completions/min_length": 694.0, |
| "completions/min_terminated_length": 694.0, |
| "entropy": 0.23127288930118084, |
| "epoch": 0.007874468329693069, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017832156270742416, |
| "learning_rate": 1e-05, |
| "loss": -0.0371, |
| "num_tokens": 6377558.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.275546669960022, |
| "sampling/importance_sampling_ratio/mean": 1.0002614259719849, |
| "sampling/importance_sampling_ratio/min": 0.6965088248252869, |
| "sampling/sampling_logp_difference/max": 0.36167478561401367, |
| "sampling/sampling_logp_difference/mean": 0.009490367025136948, |
| "step": 137 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00013839952953276224, |
| "clip_ratio/low_min": 0.00013839952953276224, |
| "clip_ratio/region_mean": 0.00013839952953276224, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6932.0, |
| "completions/max_terminated_length": 6932.0, |
| "completions/mean_length": 4874.625, |
| "completions/mean_terminated_length": 4874.625, |
| "completions/min_length": 2478.0, |
| "completions/min_terminated_length": 2478.0, |
| "entropy": 0.6169449612498283, |
| "epoch": 0.007931946200712726, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0270911306142807, |
| "learning_rate": 1e-05, |
| "loss": -0.0552, |
| "num_tokens": 6417859.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001020431518555, |
| "sampling/importance_sampling_ratio/min": 0.5970737338066101, |
| "sampling/sampling_logp_difference/max": 0.7154791355133057, |
| "sampling/sampling_logp_difference/mean": 0.015228682197630405, |
| "step": 138 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2991.0, |
| "completions/max_terminated_length": 2991.0, |
| "completions/mean_length": 2041.875, |
| "completions/mean_terminated_length": 2041.875, |
| "completions/min_length": 1428.0, |
| "completions/min_terminated_length": 1428.0, |
| "entropy": 0.180243456736207, |
| "epoch": 0.007989424071732382, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6435058.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3439971208572388, |
| "sampling/importance_sampling_ratio/mean": 1.0000247955322266, |
| "sampling/importance_sampling_ratio/min": 0.6104732751846313, |
| "sampling/sampling_logp_difference/max": 0.49352073669433594, |
| "sampling/sampling_logp_difference/mean": 0.007695259992033243, |
| "step": 139 |
| }, |
| { |
| "clip_ratio/high_max": 7.995203668542672e-05, |
| "clip_ratio/high_mean": 7.995203668542672e-05, |
| "clip_ratio/low_mean": 0.00029105868452461436, |
| "clip_ratio/low_min": 0.00029105868452461436, |
| "clip_ratio/region_mean": 0.0003710107212100411, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5747.0, |
| "completions/max_terminated_length": 5747.0, |
| "completions/mean_length": 3175.125, |
| "completions/mean_terminated_length": 3175.125, |
| "completions/min_length": 1304.0, |
| "completions/min_terminated_length": 1304.0, |
| "entropy": 0.3940267525613308, |
| "epoch": 0.00804690194275204, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019426902756094933, |
| "learning_rate": 1e-05, |
| "loss": -0.1836, |
| "num_tokens": 6461451.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.800508737564087, |
| "sampling/importance_sampling_ratio/mean": 0.9997274875640869, |
| "sampling/importance_sampling_ratio/min": 0.4268714189529419, |
| "sampling/sampling_logp_difference/max": 0.851272463798523, |
| "sampling/sampling_logp_difference/mean": 0.016020389273762703, |
| "step": 140 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3173.0, |
| "completions/max_terminated_length": 3173.0, |
| "completions/mean_length": 2345.75, |
| "completions/mean_terminated_length": 2345.75, |
| "completions/min_length": 1536.0, |
| "completions/min_terminated_length": 1536.0, |
| "entropy": 0.19869183097034693, |
| "epoch": 0.008104379813771698, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6481257.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.636795163154602, |
| "sampling/importance_sampling_ratio/mean": 1.000019907951355, |
| "sampling/importance_sampling_ratio/min": 0.6332414746284485, |
| "sampling/sampling_logp_difference/max": 0.4927401542663574, |
| "sampling/sampling_logp_difference/mean": 0.008244763128459454, |
| "step": 141 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017878098151413724, |
| "clip_ratio/high_mean": 0.00017878098151413724, |
| "clip_ratio/low_mean": 0.0002086660679196939, |
| "clip_ratio/low_min": 0.0002086660679196939, |
| "clip_ratio/region_mean": 0.0003874470494338311, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7785.0, |
| "completions/max_terminated_length": 7785.0, |
| "completions/mean_length": 5947.625, |
| "completions/mean_terminated_length": 5947.625, |
| "completions/min_length": 3582.0, |
| "completions/min_terminated_length": 3582.0, |
| "entropy": 0.29591018334031105, |
| "epoch": 0.008161857684791355, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0072945160791277885, |
| "learning_rate": 1e-05, |
| "loss": 0.0789, |
| "num_tokens": 6529750.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997889995574951, |
| "sampling/importance_sampling_ratio/min": 0.374642550945282, |
| "sampling/sampling_logp_difference/max": 0.9817829132080078, |
| "sampling/sampling_logp_difference/mean": 0.013533972203731537, |
| "step": 142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1034.0, |
| "completions/max_terminated_length": 1034.0, |
| "completions/mean_length": 813.625, |
| "completions/mean_terminated_length": 813.625, |
| "completions/min_length": 532.0, |
| "completions/min_terminated_length": 532.0, |
| "entropy": 0.557020515203476, |
| "epoch": 0.008219335555811013, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6538291.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2654404640197754, |
| "sampling/importance_sampling_ratio/mean": 1.0006256103515625, |
| "sampling/importance_sampling_ratio/min": 0.7822061777114868, |
| "sampling/sampling_logp_difference/max": 0.2456369400024414, |
| "sampling/sampling_logp_difference/mean": 0.015856806188821793, |
| "step": 143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3080.0, |
| "completions/max_terminated_length": 3080.0, |
| "completions/mean_length": 1541.0, |
| "completions/mean_terminated_length": 1541.0, |
| "completions/min_length": 913.0, |
| "completions/min_terminated_length": 913.0, |
| "entropy": 0.2369413785636425, |
| "epoch": 0.00827681342683067, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009322436526417732, |
| "learning_rate": 1e-05, |
| "loss": 0.3527, |
| "num_tokens": 6551547.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3686444759368896, |
| "sampling/importance_sampling_ratio/mean": 1.000044822692871, |
| "sampling/importance_sampling_ratio/min": 0.7421360015869141, |
| "sampling/sampling_logp_difference/max": 0.31382083892822266, |
| "sampling/sampling_logp_difference/mean": 0.009175000712275505, |
| "step": 144 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6132.0, |
| "completions/max_terminated_length": 6132.0, |
| "completions/mean_length": 3661.75, |
| "completions/mean_terminated_length": 3661.75, |
| "completions/min_length": 1555.0, |
| "completions/min_terminated_length": 1555.0, |
| "entropy": 0.41266872733831406, |
| "epoch": 0.008334291297850328, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6581657.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6183322668075562, |
| "sampling/importance_sampling_ratio/mean": 1.000028133392334, |
| "sampling/importance_sampling_ratio/min": 0.648280143737793, |
| "sampling/sampling_logp_difference/max": 0.4813961982727051, |
| "sampling/sampling_logp_difference/mean": 0.01620025560259819, |
| "step": 145 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013272334399516694, |
| "clip_ratio/high_mean": 0.00013272334399516694, |
| "clip_ratio/low_mean": 0.00012308468103583436, |
| "clip_ratio/low_min": 0.00012308468103583436, |
| "clip_ratio/region_mean": 0.0002558080250310013, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14048.0, |
| "completions/max_terminated_length": 14048.0, |
| "completions/mean_length": 10919.5, |
| "completions/mean_terminated_length": 10919.5, |
| "completions/min_length": 8052.0, |
| "completions/min_terminated_length": 8052.0, |
| "entropy": 0.46391235664486885, |
| "epoch": 0.008391769168869984, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011609341949224472, |
| "learning_rate": 1e-05, |
| "loss": -0.0312, |
| "num_tokens": 6670045.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998831748962402, |
| "sampling/importance_sampling_ratio/min": 0.267840713262558, |
| "sampling/sampling_logp_difference/max": 1.3173627853393555, |
| "sampling/sampling_logp_difference/mean": 0.019570058211684227, |
| "step": 146 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002113405971613247, |
| "clip_ratio/high_mean": 0.0002113405971613247, |
| "clip_ratio/low_mean": 0.00013204225979279727, |
| "clip_ratio/low_min": 0.00013204225979279727, |
| "clip_ratio/region_mean": 0.00034338285695412196, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3221.0, |
| "completions/max_terminated_length": 3221.0, |
| "completions/mean_length": 2172.5, |
| "completions/mean_terminated_length": 2172.5, |
| "completions/min_length": 894.0, |
| "completions/min_terminated_length": 894.0, |
| "entropy": 0.3832019381225109, |
| "epoch": 0.008449247039889642, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009099716320633888, |
| "learning_rate": 1e-05, |
| "loss": 0.1086, |
| "num_tokens": 6688649.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4237631559371948, |
| "sampling/importance_sampling_ratio/mean": 0.9996580481529236, |
| "sampling/importance_sampling_ratio/min": 0.6517035961151123, |
| "sampling/sampling_logp_difference/max": 0.4281654357910156, |
| "sampling/sampling_logp_difference/mean": 0.01533388253301382, |
| "step": 147 |
| }, |
| { |
| "clip_ratio/high_max": 3.787980313063599e-05, |
| "clip_ratio/high_mean": 3.787980313063599e-05, |
| "clip_ratio/low_mean": 8.50484793772921e-05, |
| "clip_ratio/low_min": 8.50484793772921e-05, |
| "clip_ratio/region_mean": 0.0001229282825079281, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11758.0, |
| "completions/max_terminated_length": 11758.0, |
| "completions/mean_length": 5206.0, |
| "completions/mean_terminated_length": 5206.0, |
| "completions/min_length": 2350.0, |
| "completions/min_terminated_length": 2350.0, |
| "entropy": 0.44523949921131134, |
| "epoch": 0.0085067249109093, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00463527487590909, |
| "learning_rate": 1e-05, |
| "loss": 0.4445, |
| "num_tokens": 6731249.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4365485906600952, |
| "sampling/importance_sampling_ratio/mean": 0.9995553493499756, |
| "sampling/importance_sampling_ratio/min": 0.6018043160438538, |
| "sampling/sampling_logp_difference/max": 0.5078229904174805, |
| "sampling/sampling_logp_difference/mean": 0.013541629537940025, |
| "step": 148 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1831.0, |
| "completions/max_terminated_length": 1831.0, |
| "completions/mean_length": 1088.25, |
| "completions/mean_terminated_length": 1088.25, |
| "completions/min_length": 860.0, |
| "completions/min_terminated_length": 860.0, |
| "entropy": 0.24122720398008823, |
| "epoch": 0.008564202781928957, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6741003.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3806281089782715, |
| "sampling/importance_sampling_ratio/mean": 1.0003173351287842, |
| "sampling/importance_sampling_ratio/min": 0.6448020339012146, |
| "sampling/sampling_logp_difference/max": 0.4388120174407959, |
| "sampling/sampling_logp_difference/mean": 0.008040092885494232, |
| "step": 149 |
| }, |
| { |
| "clip_ratio/high_max": 0.00015799710672581568, |
| "clip_ratio/high_mean": 0.00015799710672581568, |
| "clip_ratio/low_mean": 2.6057952709379606e-05, |
| "clip_ratio/low_min": 2.6057952709379606e-05, |
| "clip_ratio/region_mean": 0.0001840550594351953, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4797.0, |
| "completions/max_terminated_length": 4797.0, |
| "completions/mean_length": 2548.0, |
| "completions/mean_terminated_length": 2548.0, |
| "completions/min_length": 1179.0, |
| "completions/min_terminated_length": 1179.0, |
| "entropy": 0.4162476062774658, |
| "epoch": 0.008621680652948615, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010240177623927593, |
| "learning_rate": 1e-05, |
| "loss": 0.3125, |
| "num_tokens": 6762675.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4358079433441162, |
| "sampling/importance_sampling_ratio/mean": 1.000313401222229, |
| "sampling/importance_sampling_ratio/min": 0.6893447041511536, |
| "sampling/sampling_logp_difference/max": 0.37201380729675293, |
| "sampling/sampling_logp_difference/mean": 0.013260569423437119, |
| "step": 150 |
| }, |
| { |
| "clip_ratio/high_max": 6.30233535048319e-05, |
| "clip_ratio/high_mean": 6.30233535048319e-05, |
| "clip_ratio/low_mean": 0.00042105267857550643, |
| "clip_ratio/low_min": 0.00042105267857550643, |
| "clip_ratio/region_mean": 0.00048407603208033834, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12958.0, |
| "completions/max_terminated_length": 12958.0, |
| "completions/mean_length": 9031.875, |
| "completions/mean_terminated_length": 9031.875, |
| "completions/min_length": 4371.0, |
| "completions/min_terminated_length": 4371.0, |
| "entropy": 0.42483629286289215, |
| "epoch": 0.008679158523968273, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.029668988659977913, |
| "learning_rate": 1e-05, |
| "loss": 0.067, |
| "num_tokens": 6836258.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.7135207653045654, |
| "sampling/importance_sampling_ratio/mean": 1.0001245737075806, |
| "sampling/importance_sampling_ratio/min": 0.335127055644989, |
| "sampling/sampling_logp_difference/max": 1.093245506286621, |
| "sampling/sampling_logp_difference/mean": 0.019100628793239594, |
| "step": 151 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1469.0, |
| "completions/max_terminated_length": 1469.0, |
| "completions/mean_length": 910.0, |
| "completions/mean_terminated_length": 910.0, |
| "completions/min_length": 554.0, |
| "completions/min_terminated_length": 554.0, |
| "entropy": 0.301215760409832, |
| "epoch": 0.00873663639498793, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6844242.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.296616792678833, |
| "sampling/importance_sampling_ratio/mean": 0.9998509287834167, |
| "sampling/importance_sampling_ratio/min": 0.695597767829895, |
| "sampling/sampling_logp_difference/max": 0.36298370361328125, |
| "sampling/sampling_logp_difference/mean": 0.011990005150437355, |
| "step": 152 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1782.0, |
| "completions/max_terminated_length": 1782.0, |
| "completions/mean_length": 1431.0, |
| "completions/mean_terminated_length": 1431.0, |
| "completions/min_length": 814.0, |
| "completions/min_terminated_length": 814.0, |
| "entropy": 0.2171723898500204, |
| "epoch": 0.008794114266007588, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01422662939876318, |
| "learning_rate": 1e-05, |
| "loss": -0.0469, |
| "num_tokens": 6856698.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3222421407699585, |
| "sampling/importance_sampling_ratio/mean": 0.9996227622032166, |
| "sampling/importance_sampling_ratio/min": 0.6277742385864258, |
| "sampling/sampling_logp_difference/max": 0.4655747413635254, |
| "sampling/sampling_logp_difference/mean": 0.009093833155930042, |
| "step": 153 |
| }, |
| { |
| "clip_ratio/high_max": 0.00020458265498746186, |
| "clip_ratio/high_mean": 0.00020458265498746186, |
| "clip_ratio/low_mean": 0.000125376129290089, |
| "clip_ratio/low_min": 0.000125376129290089, |
| "clip_ratio/region_mean": 0.0003299587842775509, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1637.0, |
| "completions/max_terminated_length": 1637.0, |
| "completions/mean_length": 1273.375, |
| "completions/mean_terminated_length": 1273.375, |
| "completions/min_length": 997.0, |
| "completions/min_terminated_length": 997.0, |
| "entropy": 0.21633286029100418, |
| "epoch": 0.008851592137027244, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.059154924005270004, |
| "learning_rate": 1e-05, |
| "loss": -0.0553, |
| "num_tokens": 6867837.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.5729005336761475, |
| "sampling/importance_sampling_ratio/mean": 0.9998074769973755, |
| "sampling/importance_sampling_ratio/min": 0.6064621210098267, |
| "sampling/sampling_logp_difference/max": 0.5001130104064941, |
| "sampling/sampling_logp_difference/mean": 0.008797754533588886, |
| "step": 154 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2549.0, |
| "completions/max_terminated_length": 2549.0, |
| "completions/mean_length": 1737.375, |
| "completions/mean_terminated_length": 1737.375, |
| "completions/min_length": 976.0, |
| "completions/min_terminated_length": 976.0, |
| "entropy": 0.21732072532176971, |
| "epoch": 0.008909070008046902, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6883088.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2848201990127563, |
| "sampling/importance_sampling_ratio/mean": 1.0001574754714966, |
| "sampling/importance_sampling_ratio/min": 0.49255454540252686, |
| "sampling/sampling_logp_difference/max": 0.7081500887870789, |
| "sampling/sampling_logp_difference/mean": 0.008940266445279121, |
| "step": 155 |
| }, |
| { |
| "clip_ratio/high_max": 4.288164564059116e-05, |
| "clip_ratio/high_mean": 4.288164564059116e-05, |
| "clip_ratio/low_mean": 9.36329597607255e-05, |
| "clip_ratio/low_min": 9.36329597607255e-05, |
| "clip_ratio/region_mean": 0.00013651460540131666, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2915.0, |
| "completions/max_terminated_length": 2915.0, |
| "completions/mean_length": 1733.0, |
| "completions/mean_terminated_length": 1733.0, |
| "completions/min_length": 743.0, |
| "completions/min_terminated_length": 743.0, |
| "entropy": 0.2394898571074009, |
| "epoch": 0.00896654787906656, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.06945914030075073, |
| "learning_rate": 1e-05, |
| "loss": -0.0806, |
| "num_tokens": 6898088.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3556156158447266, |
| "sampling/importance_sampling_ratio/mean": 1.0001763105392456, |
| "sampling/importance_sampling_ratio/min": 0.6855784058570862, |
| "sampling/sampling_logp_difference/max": 0.37749242782592773, |
| "sampling/sampling_logp_difference/mean": 0.010261802934110165, |
| "step": 156 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6742.0, |
| "completions/max_terminated_length": 6742.0, |
| "completions/mean_length": 3985.5, |
| "completions/mean_terminated_length": 3985.5, |
| "completions/min_length": 1301.0, |
| "completions/min_terminated_length": 1301.0, |
| "entropy": 0.3295823559165001, |
| "epoch": 0.009024025750086217, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6931212.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3633723258972168, |
| "sampling/importance_sampling_ratio/mean": 0.9998742341995239, |
| "sampling/importance_sampling_ratio/min": 0.6257620453834534, |
| "sampling/sampling_logp_difference/max": 0.4687851071357727, |
| "sampling/sampling_logp_difference/mean": 0.014020584523677826, |
| "step": 157 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6554.0, |
| "completions/max_terminated_length": 6554.0, |
| "completions/mean_length": 3004.375, |
| "completions/mean_terminated_length": 3004.375, |
| "completions/min_length": 1153.0, |
| "completions/min_terminated_length": 1153.0, |
| "entropy": 0.2464207522571087, |
| "epoch": 0.009081503621105875, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 6955927.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4352158308029175, |
| "sampling/importance_sampling_ratio/mean": 0.9999135732650757, |
| "sampling/importance_sampling_ratio/min": 0.6133259534835815, |
| "sampling/sampling_logp_difference/max": 0.488858699798584, |
| "sampling/sampling_logp_difference/mean": 0.01061803288757801, |
| "step": 158 |
| }, |
| { |
| "clip_ratio/high_max": 3.569319142116001e-05, |
| "clip_ratio/high_mean": 3.569319142116001e-05, |
| "clip_ratio/low_mean": 0.00049591064453125, |
| "clip_ratio/low_min": 0.00049591064453125, |
| "clip_ratio/region_mean": 0.00053160383595241, |
| "completions/clipped_ratio": 0.75, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14928.0, |
| "completions/mean_length": 15631.875, |
| "completions/mean_terminated_length": 13375.5, |
| "completions/min_length": 11823.0, |
| "completions/min_terminated_length": 11823.0, |
| "entropy": 0.3680622801184654, |
| "epoch": 0.009138981492125532, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017388461157679558, |
| "learning_rate": 1e-05, |
| "loss": 0.0778, |
| "num_tokens": 7081950.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999274611473083, |
| "sampling/importance_sampling_ratio/min": 0.18218164145946503, |
| "sampling/sampling_logp_difference/max": 1.7027510404586792, |
| "sampling/sampling_logp_difference/mean": 0.01629018411040306, |
| "step": 159 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4961.0, |
| "completions/max_terminated_length": 4961.0, |
| "completions/mean_length": 4246.625, |
| "completions/mean_terminated_length": 4246.625, |
| "completions/min_length": 3206.0, |
| "completions/min_terminated_length": 3206.0, |
| "entropy": 0.3007093593478203, |
| "epoch": 0.00919645936314519, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7116963.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4973028898239136, |
| "sampling/importance_sampling_ratio/mean": 1.000224232673645, |
| "sampling/importance_sampling_ratio/min": 0.6020798683166504, |
| "sampling/sampling_logp_difference/max": 0.5073652267456055, |
| "sampling/sampling_logp_difference/mean": 0.011745051480829716, |
| "step": 160 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14000.0, |
| "completions/max_terminated_length": 14000.0, |
| "completions/mean_length": 6351.75, |
| "completions/mean_terminated_length": 6351.75, |
| "completions/min_length": 1756.0, |
| "completions/min_terminated_length": 1756.0, |
| "entropy": 0.5105179361999035, |
| "epoch": 0.009253937234164846, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7168593.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000451803207397, |
| "sampling/importance_sampling_ratio/min": 0.07039714604616165, |
| "sampling/sampling_logp_difference/max": 2.6536026000976562, |
| "sampling/sampling_logp_difference/mean": 0.021646125242114067, |
| "step": 161 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8577.0, |
| "completions/max_terminated_length": 8577.0, |
| "completions/mean_length": 4887.25, |
| "completions/mean_terminated_length": 4887.25, |
| "completions/min_length": 2752.0, |
| "completions/min_terminated_length": 2752.0, |
| "entropy": 0.43511849269270897, |
| "epoch": 0.009311415105184503, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7208419.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4992823600769043, |
| "sampling/importance_sampling_ratio/mean": 0.9998119473457336, |
| "sampling/importance_sampling_ratio/min": 0.5442285537719727, |
| "sampling/sampling_logp_difference/max": 0.6083860397338867, |
| "sampling/sampling_logp_difference/mean": 0.016671858727931976, |
| "step": 162 |
| }, |
| { |
| "clip_ratio/high_max": 0.00022961249851505272, |
| "clip_ratio/high_mean": 0.00022961249851505272, |
| "clip_ratio/low_mean": 0.0002116431569447741, |
| "clip_ratio/low_min": 0.0002116431569447741, |
| "clip_ratio/region_mean": 0.0004412556554598268, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12969.0, |
| "completions/max_terminated_length": 12969.0, |
| "completions/mean_length": 9163.5, |
| "completions/mean_terminated_length": 9163.5, |
| "completions/min_length": 4514.0, |
| "completions/min_terminated_length": 4514.0, |
| "entropy": 0.49257949367165565, |
| "epoch": 0.009368892976204161, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006566404365003109, |
| "learning_rate": 1e-05, |
| "loss": 0.0194, |
| "num_tokens": 7282679.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998836517333984, |
| "sampling/importance_sampling_ratio/min": 0.3657211363315582, |
| "sampling/sampling_logp_difference/max": 1.0058841705322266, |
| "sampling/sampling_logp_difference/mean": 0.020654143765568733, |
| "step": 163 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16245.0, |
| "completions/mean_length": 11264.25, |
| "completions/mean_terminated_length": 10532.857421875, |
| "completions/min_length": 6847.0, |
| "completions/min_terminated_length": 6847.0, |
| "entropy": 0.7053465247154236, |
| "epoch": 0.009426370847223819, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7374009.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000031590461731, |
| "sampling/importance_sampling_ratio/min": 0.302106112241745, |
| "sampling/sampling_logp_difference/max": 1.3456833362579346, |
| "sampling/sampling_logp_difference/mean": 0.02689768746495247, |
| "step": 164 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1544.0, |
| "completions/max_terminated_length": 1544.0, |
| "completions/mean_length": 1000.25, |
| "completions/mean_terminated_length": 1000.25, |
| "completions/min_length": 640.0, |
| "completions/min_terminated_length": 640.0, |
| "entropy": 0.6362258717417717, |
| "epoch": 0.009483848718243476, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7382827.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2800841331481934, |
| "sampling/importance_sampling_ratio/mean": 1.00029456615448, |
| "sampling/importance_sampling_ratio/min": 0.7599013447761536, |
| "sampling/sampling_logp_difference/max": 0.274566650390625, |
| "sampling/sampling_logp_difference/mean": 0.017346898093819618, |
| "step": 165 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012542009426397271, |
| "clip_ratio/high_mean": 0.00012542009426397271, |
| "clip_ratio/low_mean": 0.00020808458793908358, |
| "clip_ratio/low_min": 0.00020808458793908358, |
| "clip_ratio/region_mean": 0.0003335046822030563, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6953.0, |
| "completions/max_terminated_length": 6953.0, |
| "completions/mean_length": 3977.5, |
| "completions/mean_terminated_length": 3977.5, |
| "completions/min_length": 2419.0, |
| "completions/min_terminated_length": 2419.0, |
| "entropy": 0.2975945267826319, |
| "epoch": 0.009541326589263134, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0099942646920681, |
| "learning_rate": 1e-05, |
| "loss": 0.1789, |
| "num_tokens": 7416031.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8197044134140015, |
| "sampling/importance_sampling_ratio/mean": 1.0002262592315674, |
| "sampling/importance_sampling_ratio/min": 0.6547252535820007, |
| "sampling/sampling_logp_difference/max": 0.5986740589141846, |
| "sampling/sampling_logp_difference/mean": 0.01162185799330473, |
| "step": 166 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14021.0, |
| "completions/mean_length": 7496.625, |
| "completions/mean_terminated_length": 4534.1669921875, |
| "completions/min_length": 1406.0, |
| "completions/min_terminated_length": 1406.0, |
| "entropy": 0.27918735332787037, |
| "epoch": 0.009598804460282792, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7477228.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998542666435242, |
| "sampling/importance_sampling_ratio/min": 0.4236017167568207, |
| "sampling/sampling_logp_difference/max": 0.8589615821838379, |
| "sampling/sampling_logp_difference/mean": 0.014311072416603565, |
| "step": 167 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1618.0, |
| "completions/max_terminated_length": 1618.0, |
| "completions/mean_length": 661.25, |
| "completions/mean_terminated_length": 661.25, |
| "completions/min_length": 353.0, |
| "completions/min_terminated_length": 353.0, |
| "entropy": 0.20153271220624447, |
| "epoch": 0.00965628233130245, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7483510.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3847815990447998, |
| "sampling/importance_sampling_ratio/mean": 0.9990352392196655, |
| "sampling/importance_sampling_ratio/min": 0.6852436065673828, |
| "sampling/sampling_logp_difference/max": 0.3779808282852173, |
| "sampling/sampling_logp_difference/mean": 0.009825999848544598, |
| "step": 168 |
| }, |
| { |
| "clip_ratio/high_max": 6.28524212515913e-05, |
| "clip_ratio/high_mean": 6.28524212515913e-05, |
| "clip_ratio/low_mean": 0.0007833648342057131, |
| "clip_ratio/low_min": 0.0007833648342057131, |
| "clip_ratio/region_mean": 0.0008462172554573044, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6571.0, |
| "completions/max_terminated_length": 6571.0, |
| "completions/mean_length": 4029.25, |
| "completions/mean_terminated_length": 4029.25, |
| "completions/min_length": 950.0, |
| "completions/min_terminated_length": 950.0, |
| "entropy": 0.3966398164629936, |
| "epoch": 0.009713760202322105, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03468601778149605, |
| "learning_rate": 1e-05, |
| "loss": 0.0808, |
| "num_tokens": 7517736.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4819731712341309, |
| "sampling/importance_sampling_ratio/mean": 0.9999225735664368, |
| "sampling/importance_sampling_ratio/min": 0.4870922863483429, |
| "sampling/sampling_logp_difference/max": 0.719301700592041, |
| "sampling/sampling_logp_difference/mean": 0.01768554002046585, |
| "step": 169 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2629.0, |
| "completions/max_terminated_length": 2629.0, |
| "completions/mean_length": 2078.125, |
| "completions/mean_terminated_length": 2078.125, |
| "completions/min_length": 1474.0, |
| "completions/min_terminated_length": 1474.0, |
| "entropy": 0.25761128030717373, |
| "epoch": 0.009771238073341763, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7536521.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3825703859329224, |
| "sampling/importance_sampling_ratio/mean": 0.9997692108154297, |
| "sampling/importance_sampling_ratio/min": 0.6662405729293823, |
| "sampling/sampling_logp_difference/max": 0.4061044454574585, |
| "sampling/sampling_logp_difference/mean": 0.01018275786191225, |
| "step": 170 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4610.0, |
| "completions/max_terminated_length": 4610.0, |
| "completions/mean_length": 2941.875, |
| "completions/mean_terminated_length": 2941.875, |
| "completions/min_length": 1798.0, |
| "completions/min_terminated_length": 1798.0, |
| "entropy": 0.7922124117612839, |
| "epoch": 0.00982871594436142, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7561912.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.655548334121704, |
| "sampling/importance_sampling_ratio/mean": 1.000400424003601, |
| "sampling/importance_sampling_ratio/min": 0.6684586405754089, |
| "sampling/sampling_logp_difference/max": 0.5041322708129883, |
| "sampling/sampling_logp_difference/mean": 0.019063184037804604, |
| "step": 171 |
| }, |
| { |
| "clip_ratio/high_max": 6.214732457010541e-05, |
| "clip_ratio/high_mean": 6.214732457010541e-05, |
| "clip_ratio/low_mean": 0.00022542186343343928, |
| "clip_ratio/low_min": 0.00022542186343343928, |
| "clip_ratio/region_mean": 0.0002875691880035447, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11616.0, |
| "completions/max_terminated_length": 11616.0, |
| "completions/mean_length": 6968.0, |
| "completions/mean_terminated_length": 6968.0, |
| "completions/min_length": 4674.0, |
| "completions/min_terminated_length": 4674.0, |
| "entropy": 0.3691304251551628, |
| "epoch": 0.009886193815381078, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.043332166969776154, |
| "learning_rate": 1e-05, |
| "loss": 0.2118, |
| "num_tokens": 7618896.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8682059049606323, |
| "sampling/importance_sampling_ratio/mean": 1.000428318977356, |
| "sampling/importance_sampling_ratio/min": 0.4910013973712921, |
| "sampling/sampling_logp_difference/max": 0.7113083600997925, |
| "sampling/sampling_logp_difference/mean": 0.015282669104635715, |
| "step": 172 |
| }, |
| { |
| "clip_ratio/high_max": 0.00016217795018746983, |
| "clip_ratio/high_mean": 0.00016217795018746983, |
| "clip_ratio/low_mean": 0.00034929594403365627, |
| "clip_ratio/low_min": 0.00034929594403365627, |
| "clip_ratio/region_mean": 0.0005114738942211261, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15481.0, |
| "completions/max_terminated_length": 15481.0, |
| "completions/mean_length": 9216.875, |
| "completions/mean_terminated_length": 9216.875, |
| "completions/min_length": 3655.0, |
| "completions/min_terminated_length": 3655.0, |
| "entropy": 0.6708216592669487, |
| "epoch": 0.009943671686400736, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.030088379979133606, |
| "learning_rate": 1e-05, |
| "loss": 0.1355, |
| "num_tokens": 7693511.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.8613359928131104, |
| "sampling/importance_sampling_ratio/mean": 0.999386191368103, |
| "sampling/importance_sampling_ratio/min": 0.4131677448749542, |
| "sampling/sampling_logp_difference/max": 0.8839015960693359, |
| "sampling/sampling_logp_difference/mean": 0.027095327153801918, |
| "step": 173 |
| }, |
| { |
| "clip_ratio/high_max": 5.058680835645646e-05, |
| "clip_ratio/high_mean": 5.058680835645646e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 5.058680835645646e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4366.0, |
| "completions/max_terminated_length": 4366.0, |
| "completions/mean_length": 1801.875, |
| "completions/mean_terminated_length": 1801.875, |
| "completions/min_length": 610.0, |
| "completions/min_terminated_length": 610.0, |
| "entropy": 0.24556374922394753, |
| "epoch": 0.010001149557420394, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011647317558526993, |
| "learning_rate": 1e-05, |
| "loss": -0.1987, |
| "num_tokens": 7709054.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6970995664596558, |
| "sampling/importance_sampling_ratio/mean": 0.9994653463363647, |
| "sampling/importance_sampling_ratio/min": 0.6818316578865051, |
| "sampling/sampling_logp_difference/max": 0.5289206504821777, |
| "sampling/sampling_logp_difference/mean": 0.010972586460411549, |
| "step": 174 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013646288425661623, |
| "clip_ratio/high_mean": 0.00013646288425661623, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.00013646288425661623, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1411.0, |
| "completions/max_terminated_length": 1411.0, |
| "completions/mean_length": 850.875, |
| "completions/mean_terminated_length": 850.875, |
| "completions/min_length": 528.0, |
| "completions/min_terminated_length": 528.0, |
| "entropy": 0.28919071331620216, |
| "epoch": 0.010058627428440051, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.10567350685596466, |
| "learning_rate": 1e-05, |
| "loss": -0.0169, |
| "num_tokens": 7716869.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002267360687256, |
| "sampling/importance_sampling_ratio/min": 0.6456519365310669, |
| "sampling/sampling_logp_difference/max": 0.8261299133300781, |
| "sampling/sampling_logp_difference/mean": 0.012845244258642197, |
| "step": 175 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5941.0, |
| "completions/max_terminated_length": 5941.0, |
| "completions/mean_length": 3116.375, |
| "completions/mean_terminated_length": 3116.375, |
| "completions/min_length": 1754.0, |
| "completions/min_terminated_length": 1754.0, |
| "entropy": 0.264639051631093, |
| "epoch": 0.010116105299459707, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7742856.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3892130851745605, |
| "sampling/importance_sampling_ratio/mean": 1.0001221895217896, |
| "sampling/importance_sampling_ratio/min": 0.5892245769500732, |
| "sampling/sampling_logp_difference/max": 0.5289478302001953, |
| "sampling/sampling_logp_difference/mean": 0.010956798680126667, |
| "step": 176 |
| }, |
| { |
| "clip_ratio/high_max": 7.992327300598845e-05, |
| "clip_ratio/high_mean": 7.992327300598845e-05, |
| "clip_ratio/low_mean": 0.00014164306048769504, |
| "clip_ratio/low_min": 0.00014164306048769504, |
| "clip_ratio/region_mean": 0.0002215663334936835, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1765.0, |
| "completions/max_terminated_length": 1765.0, |
| "completions/mean_length": 1319.75, |
| "completions/mean_terminated_length": 1319.75, |
| "completions/min_length": 698.0, |
| "completions/min_terminated_length": 698.0, |
| "entropy": 0.24970881827175617, |
| "epoch": 0.010173583170479365, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016001326963305473, |
| "learning_rate": 1e-05, |
| "loss": 0.1191, |
| "num_tokens": 7754102.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.398911952972412, |
| "sampling/importance_sampling_ratio/mean": 1.0002716779708862, |
| "sampling/importance_sampling_ratio/min": 0.6763787269592285, |
| "sampling/sampling_logp_difference/max": 0.3910020589828491, |
| "sampling/sampling_logp_difference/mean": 0.010832153260707855, |
| "step": 177 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10050.0, |
| "completions/max_terminated_length": 10050.0, |
| "completions/mean_length": 4924.625, |
| "completions/mean_terminated_length": 4924.625, |
| "completions/min_length": 1384.0, |
| "completions/min_terminated_length": 1384.0, |
| "entropy": 0.39915452152490616, |
| "epoch": 0.010231061041499023, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7795035.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003193616867065, |
| "sampling/importance_sampling_ratio/min": 0.4675396680831909, |
| "sampling/sampling_logp_difference/max": 1.1266989707946777, |
| "sampling/sampling_logp_difference/mean": 0.01402293797582388, |
| "step": 178 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014332833416119684, |
| "clip_ratio/high_mean": 0.00014332833416119684, |
| "clip_ratio/low_mean": 0.0005741491113440134, |
| "clip_ratio/low_min": 0.0005741491113440134, |
| "clip_ratio/region_mean": 0.0007174774455052102, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6375.0, |
| "completions/max_terminated_length": 6375.0, |
| "completions/mean_length": 3390.625, |
| "completions/mean_terminated_length": 3390.625, |
| "completions/min_length": 559.0, |
| "completions/min_terminated_length": 559.0, |
| "entropy": 0.6463589556515217, |
| "epoch": 0.01028853891251868, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024347297847270966, |
| "learning_rate": 1e-05, |
| "loss": -0.2196, |
| "num_tokens": 7826600.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.6255055665969849, |
| "sampling/importance_sampling_ratio/mean": 0.9999240040779114, |
| "sampling/importance_sampling_ratio/min": 0.30804312229156494, |
| "sampling/sampling_logp_difference/max": 1.1775155067443848, |
| "sampling/sampling_logp_difference/mean": 0.026130210608243942, |
| "step": 179 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5437.0, |
| "completions/max_terminated_length": 5437.0, |
| "completions/mean_length": 2589.5, |
| "completions/mean_terminated_length": 2589.5, |
| "completions/min_length": 1455.0, |
| "completions/min_terminated_length": 1455.0, |
| "entropy": 0.4448091350495815, |
| "epoch": 0.010346016783538338, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7848316.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5598030090332031, |
| "sampling/importance_sampling_ratio/mean": 1.0002201795578003, |
| "sampling/importance_sampling_ratio/min": 0.6510143280029297, |
| "sampling/sampling_logp_difference/max": 0.44455957412719727, |
| "sampling/sampling_logp_difference/mean": 0.01733734831213951, |
| "step": 180 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15770.0, |
| "completions/max_terminated_length": 15770.0, |
| "completions/mean_length": 11755.25, |
| "completions/mean_terminated_length": 11755.25, |
| "completions/min_length": 5009.0, |
| "completions/min_terminated_length": 5009.0, |
| "entropy": 0.524201687425375, |
| "epoch": 0.010403494654557996, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7943358.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999759793281555, |
| "sampling/importance_sampling_ratio/min": 0.17881280183792114, |
| "sampling/sampling_logp_difference/max": 1.7214158773422241, |
| "sampling/sampling_logp_difference/mean": 0.022415822371840477, |
| "step": 181 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002902467967942357, |
| "clip_ratio/high_mean": 0.0002902467967942357, |
| "clip_ratio/low_mean": 0.00016747052723076195, |
| "clip_ratio/low_min": 0.00016747052723076195, |
| "clip_ratio/region_mean": 0.00045771732402499765, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11196.0, |
| "completions/max_terminated_length": 11196.0, |
| "completions/mean_length": 3628.25, |
| "completions/mean_terminated_length": 3628.25, |
| "completions/min_length": 1508.0, |
| "completions/min_terminated_length": 1508.0, |
| "entropy": 0.6214040480554104, |
| "epoch": 0.010460972525577653, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.17145849764347076, |
| "learning_rate": 1e-05, |
| "loss": 0.7378, |
| "num_tokens": 7973424.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6428632736206055, |
| "sampling/importance_sampling_ratio/mean": 0.9999727606773376, |
| "sampling/importance_sampling_ratio/min": 0.31693235039711, |
| "sampling/sampling_logp_difference/max": 1.1490669250488281, |
| "sampling/sampling_logp_difference/mean": 0.02465110458433628, |
| "step": 182 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3473.0, |
| "completions/max_terminated_length": 3473.0, |
| "completions/mean_length": 2117.75, |
| "completions/mean_terminated_length": 2117.75, |
| "completions/min_length": 1016.0, |
| "completions/min_terminated_length": 1016.0, |
| "entropy": 0.42756710946559906, |
| "epoch": 0.01051845039659731, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 7991398.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4277199506759644, |
| "sampling/importance_sampling_ratio/mean": 0.9998186230659485, |
| "sampling/importance_sampling_ratio/min": 0.6188682317733765, |
| "sampling/sampling_logp_difference/max": 0.47986292839050293, |
| "sampling/sampling_logp_difference/mean": 0.016678232699632645, |
| "step": 183 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9563.0, |
| "completions/max_terminated_length": 9563.0, |
| "completions/mean_length": 5603.75, |
| "completions/mean_terminated_length": 5603.75, |
| "completions/min_length": 2811.0, |
| "completions/min_terminated_length": 2811.0, |
| "entropy": 0.2840829510241747, |
| "epoch": 0.010575928267616967, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8037348.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8708668947219849, |
| "sampling/importance_sampling_ratio/mean": 1.0001227855682373, |
| "sampling/importance_sampling_ratio/min": 0.36112356185913086, |
| "sampling/sampling_logp_difference/max": 1.0185351371765137, |
| "sampling/sampling_logp_difference/mean": 0.012249253690242767, |
| "step": 184 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0004523384304775391, |
| "clip_ratio/low_min": 0.0004523384304775391, |
| "clip_ratio/region_mean": 0.0004523384304775391, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11455.0, |
| "completions/max_terminated_length": 11455.0, |
| "completions/mean_length": 7349.875, |
| "completions/mean_terminated_length": 7349.875, |
| "completions/min_length": 3444.0, |
| "completions/min_terminated_length": 3444.0, |
| "entropy": 0.34330933913588524, |
| "epoch": 0.010633406138636625, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00419212132692337, |
| "learning_rate": 1e-05, |
| "loss": -0.0421, |
| "num_tokens": 8096979.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999867677688599, |
| "sampling/importance_sampling_ratio/min": 0.5320678353309631, |
| "sampling/sampling_logp_difference/max": 0.8504753112792969, |
| "sampling/sampling_logp_difference/mean": 0.014090518467128277, |
| "step": 185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3496.0, |
| "completions/max_terminated_length": 3496.0, |
| "completions/mean_length": 2670.375, |
| "completions/mean_terminated_length": 2670.375, |
| "completions/min_length": 2176.0, |
| "completions/min_terminated_length": 2176.0, |
| "entropy": 0.25105168111622334, |
| "epoch": 0.010690884009656282, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8119302.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3162728548049927, |
| "sampling/importance_sampling_ratio/mean": 1.0000258684158325, |
| "sampling/importance_sampling_ratio/min": 0.5002357959747314, |
| "sampling/sampling_logp_difference/max": 0.692675769329071, |
| "sampling/sampling_logp_difference/mean": 0.00980860274285078, |
| "step": 186 |
| }, |
| { |
| "clip_ratio/high_max": 4.545454430626705e-05, |
| "clip_ratio/high_mean": 4.545454430626705e-05, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 4.545454430626705e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3995.0, |
| "completions/max_terminated_length": 3995.0, |
| "completions/mean_length": 2801.0, |
| "completions/mean_terminated_length": 2801.0, |
| "completions/min_length": 1778.0, |
| "completions/min_terminated_length": 1778.0, |
| "entropy": 0.34686275385320187, |
| "epoch": 0.01074836188067594, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07881250977516174, |
| "learning_rate": 1e-05, |
| "loss": -0.0175, |
| "num_tokens": 8142862.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.474563479423523, |
| "sampling/importance_sampling_ratio/mean": 0.9998308420181274, |
| "sampling/importance_sampling_ratio/min": 0.6655871272087097, |
| "sampling/sampling_logp_difference/max": 0.407085657119751, |
| "sampling/sampling_logp_difference/mean": 0.013679606840014458, |
| "step": 187 |
| }, |
| { |
| "clip_ratio/high_max": 9.280240919906646e-05, |
| "clip_ratio/high_mean": 9.280240919906646e-05, |
| "clip_ratio/low_mean": 0.0005474619974847883, |
| "clip_ratio/low_min": 0.0005474619974847883, |
| "clip_ratio/region_mean": 0.0006402644066838548, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13911.0, |
| "completions/max_terminated_length": 13911.0, |
| "completions/mean_length": 6992.75, |
| "completions/mean_terminated_length": 6992.75, |
| "completions/min_length": 3146.0, |
| "completions/min_terminated_length": 3146.0, |
| "entropy": 0.439095851033926, |
| "epoch": 0.010805839751695598, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014007357880473137, |
| "learning_rate": 1e-05, |
| "loss": 0.3076, |
| "num_tokens": 8199732.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.884044885635376, |
| "sampling/importance_sampling_ratio/mean": 1.000054955482483, |
| "sampling/importance_sampling_ratio/min": 0.2771124541759491, |
| "sampling/sampling_logp_difference/max": 1.2833318710327148, |
| "sampling/sampling_logp_difference/mean": 0.01880401186645031, |
| "step": 188 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017074525840143906, |
| "clip_ratio/high_mean": 0.00017074525840143906, |
| "clip_ratio/low_mean": 0.00018244860257254913, |
| "clip_ratio/low_min": 0.00018244860257254913, |
| "clip_ratio/region_mean": 0.0003531938609739882, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13772.0, |
| "completions/mean_length": 10991.0, |
| "completions/mean_terminated_length": 10220.572265625, |
| "completions/min_length": 5741.0, |
| "completions/min_terminated_length": 5741.0, |
| "entropy": 0.5909396111965179, |
| "epoch": 0.010863317622715255, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006729756481945515, |
| "learning_rate": 1e-05, |
| "loss": 0.2001, |
| "num_tokens": 8288572.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003269910812378, |
| "sampling/importance_sampling_ratio/min": 0.35161498188972473, |
| "sampling/sampling_logp_difference/max": 1.0452184677124023, |
| "sampling/sampling_logp_difference/mean": 0.024106120690703392, |
| "step": 189 |
| }, |
| { |
| "clip_ratio/high_max": 4.512167470238637e-05, |
| "clip_ratio/high_mean": 4.512167470238637e-05, |
| "clip_ratio/low_mean": 1.870417509053368e-05, |
| "clip_ratio/low_min": 1.870417509053368e-05, |
| "clip_ratio/region_mean": 6.382584979292005e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10334.0, |
| "completions/max_terminated_length": 10334.0, |
| "completions/mean_length": 7156.25, |
| "completions/mean_terminated_length": 7156.25, |
| "completions/min_length": 1146.0, |
| "completions/min_terminated_length": 1146.0, |
| "entropy": 0.5183168314397335, |
| "epoch": 0.010920795493734913, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.027234913781285286, |
| "learning_rate": 1e-05, |
| "loss": -0.2447, |
| "num_tokens": 8347710.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000864267349243, |
| "sampling/importance_sampling_ratio/min": 0.42131614685058594, |
| "sampling/sampling_logp_difference/max": 0.99310302734375, |
| "sampling/sampling_logp_difference/mean": 0.015133515931665897, |
| "step": 190 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001190261282317806, |
| "clip_ratio/high_mean": 0.0001190261282317806, |
| "clip_ratio/low_mean": 0.00020377377950353548, |
| "clip_ratio/low_min": 0.00020377377950353548, |
| "clip_ratio/region_mean": 0.0003227999077353161, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 11128.0, |
| "completions/mean_length": 8664.375, |
| "completions/mean_terminated_length": 7561.57177734375, |
| "completions/min_length": 4658.0, |
| "completions/min_terminated_length": 4658.0, |
| "entropy": 0.46362604573369026, |
| "epoch": 0.010978273364754569, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013075378723442554, |
| "learning_rate": 1e-05, |
| "loss": 0.1808, |
| "num_tokens": 8418145.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000049114227295, |
| "sampling/importance_sampling_ratio/min": 0.30332186818122864, |
| "sampling/sampling_logp_difference/max": 1.1929607391357422, |
| "sampling/sampling_logp_difference/mean": 0.019484898075461388, |
| "step": 191 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5233.0, |
| "completions/max_terminated_length": 5233.0, |
| "completions/mean_length": 2935.5, |
| "completions/mean_terminated_length": 2935.5, |
| "completions/min_length": 2037.0, |
| "completions/min_terminated_length": 2037.0, |
| "entropy": 0.20832776837050915, |
| "epoch": 0.011035751235774226, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8442229.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3631702661514282, |
| "sampling/importance_sampling_ratio/mean": 0.9995654225349426, |
| "sampling/importance_sampling_ratio/min": 0.647485077381134, |
| "sampling/sampling_logp_difference/max": 0.434659481048584, |
| "sampling/sampling_logp_difference/mean": 0.008618641644716263, |
| "step": 192 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0011076702794525772, |
| "clip_ratio/low_min": 0.0011076702794525772, |
| "clip_ratio/region_mean": 0.0011076702794525772, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12091.0, |
| "completions/max_terminated_length": 12091.0, |
| "completions/mean_length": 9176.5, |
| "completions/mean_terminated_length": 9176.5, |
| "completions/min_length": 7268.0, |
| "completions/min_terminated_length": 7268.0, |
| "entropy": 0.8082810416817665, |
| "epoch": 0.011093229106793884, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009606908075511456, |
| "learning_rate": 1e-05, |
| "loss": 0.0418, |
| "num_tokens": 8516945.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999254941940308, |
| "sampling/importance_sampling_ratio/min": 0.24914436042308807, |
| "sampling/sampling_logp_difference/max": 1.3897228240966797, |
| "sampling/sampling_logp_difference/mean": 0.03110113926231861, |
| "step": 193 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001240125511685619, |
| "clip_ratio/high_mean": 0.0001240125511685619, |
| "clip_ratio/low_mean": 0.0002643555781105533, |
| "clip_ratio/low_min": 0.0002643555781105533, |
| "clip_ratio/region_mean": 0.00038836812927911524, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9104.0, |
| "completions/max_terminated_length": 9104.0, |
| "completions/mean_length": 6047.625, |
| "completions/mean_terminated_length": 6047.625, |
| "completions/min_length": 2831.0, |
| "completions/min_terminated_length": 2831.0, |
| "entropy": 0.7152147218585014, |
| "epoch": 0.011150706977813542, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01703065074980259, |
| "learning_rate": 1e-05, |
| "loss": 0.0503, |
| "num_tokens": 8566526.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8028453588485718, |
| "sampling/importance_sampling_ratio/mean": 1.000258207321167, |
| "sampling/importance_sampling_ratio/min": 0.4833535850048065, |
| "sampling/sampling_logp_difference/max": 0.7270069122314453, |
| "sampling/sampling_logp_difference/mean": 0.02330091968178749, |
| "step": 194 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14530.0, |
| "completions/mean_length": 14316.125, |
| "completions/mean_terminated_length": 13626.833984375, |
| "completions/min_length": 12473.0, |
| "completions/min_terminated_length": 12473.0, |
| "entropy": 0.5538299642503262, |
| "epoch": 0.0112081848488332, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8684231.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998871684074402, |
| "sampling/importance_sampling_ratio/min": 0.3882474899291992, |
| "sampling/sampling_logp_difference/max": 1.178421974182129, |
| "sampling/sampling_logp_difference/mean": 0.021088674664497375, |
| "step": 195 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9680.0, |
| "completions/max_terminated_length": 9680.0, |
| "completions/mean_length": 3682.5, |
| "completions/mean_terminated_length": 3682.5, |
| "completions/min_length": 1100.0, |
| "completions/min_terminated_length": 1100.0, |
| "entropy": 0.3728352524340153, |
| "epoch": 0.011265662719852857, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8714371.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0004141330718994, |
| "sampling/importance_sampling_ratio/min": 0.5050770044326782, |
| "sampling/sampling_logp_difference/max": 0.8590295314788818, |
| "sampling/sampling_logp_difference/mean": 0.017004601657390594, |
| "step": 196 |
| }, |
| { |
| "clip_ratio/high_max": 8.552064900868572e-05, |
| "clip_ratio/high_mean": 8.552064900868572e-05, |
| "clip_ratio/low_mean": 0.0002187505378969945, |
| "clip_ratio/low_min": 0.0002187505378969945, |
| "clip_ratio/region_mean": 0.0003042711869056802, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15636.0, |
| "completions/mean_length": 12618.375, |
| "completions/mean_terminated_length": 12080.4287109375, |
| "completions/min_length": 7212.0, |
| "completions/min_terminated_length": 7212.0, |
| "entropy": 0.558051310479641, |
| "epoch": 0.011323140590872515, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006278068292886019, |
| "learning_rate": 1e-05, |
| "loss": 0.1062, |
| "num_tokens": 8816014.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999608993530273, |
| "sampling/importance_sampling_ratio/min": 0.06352178007364273, |
| "sampling/sampling_logp_difference/max": 2.7563724517822266, |
| "sampling/sampling_logp_difference/mean": 0.020380595698952675, |
| "step": 197 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0008949943621701095, |
| "clip_ratio/low_min": 0.0008949943621701095, |
| "clip_ratio/region_mean": 0.0008949943621701095, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9994.0, |
| "completions/max_terminated_length": 9994.0, |
| "completions/mean_length": 4796.375, |
| "completions/mean_terminated_length": 4796.375, |
| "completions/min_length": 941.0, |
| "completions/min_terminated_length": 941.0, |
| "entropy": 0.5178602710366249, |
| "epoch": 0.01138061846189217, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014132932759821415, |
| "learning_rate": 1e-05, |
| "loss": -0.0059, |
| "num_tokens": 8855321.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000486373901367, |
| "sampling/importance_sampling_ratio/min": 0.32621052861213684, |
| "sampling/sampling_logp_difference/max": 1.1202123165130615, |
| "sampling/sampling_logp_difference/mean": 0.02108672820031643, |
| "step": 198 |
| }, |
| { |
| "clip_ratio/high_max": 0.00011427201388869435, |
| "clip_ratio/high_mean": 0.00011427201388869435, |
| "clip_ratio/low_mean": 9.918212890625e-05, |
| "clip_ratio/low_min": 9.918212890625e-05, |
| "clip_ratio/region_mean": 0.00021345414279494435, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13635.0, |
| "completions/mean_length": 8562.75, |
| "completions/mean_terminated_length": 7445.4287109375, |
| "completions/min_length": 1630.0, |
| "completions/min_terminated_length": 1630.0, |
| "entropy": 0.47923652827739716, |
| "epoch": 0.011438096332911828, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009647444821894169, |
| "learning_rate": 1e-05, |
| "loss": 0.3229, |
| "num_tokens": 8925015.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001194477081299, |
| "sampling/importance_sampling_ratio/min": 0.4713723063468933, |
| "sampling/sampling_logp_difference/max": 0.9788780212402344, |
| "sampling/sampling_logp_difference/mean": 0.022269317880272865, |
| "step": 199 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8952.0, |
| "completions/max_terminated_length": 8952.0, |
| "completions/mean_length": 5280.125, |
| "completions/mean_terminated_length": 5280.125, |
| "completions/min_length": 1581.0, |
| "completions/min_terminated_length": 1581.0, |
| "entropy": 0.6958541795611382, |
| "epoch": 0.011495574203931486, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 8968592.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8806509971618652, |
| "sampling/importance_sampling_ratio/mean": 0.9999502897262573, |
| "sampling/importance_sampling_ratio/min": 0.4117402136325836, |
| "sampling/sampling_logp_difference/max": 0.8873627185821533, |
| "sampling/sampling_logp_difference/mean": 0.020331639796495438, |
| "step": 200 |
| }, |
| { |
| "clip_ratio/high_max": 9.95652808342129e-05, |
| "clip_ratio/high_mean": 9.95652808342129e-05, |
| "clip_ratio/low_mean": 8.650519157527015e-05, |
| "clip_ratio/low_min": 8.650519157527015e-05, |
| "clip_ratio/region_mean": 0.00018607047240948305, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13644.0, |
| "completions/max_terminated_length": 13644.0, |
| "completions/mean_length": 7140.375, |
| "completions/mean_terminated_length": 7140.375, |
| "completions/min_length": 4784.0, |
| "completions/min_terminated_length": 4784.0, |
| "entropy": 0.5029181391000748, |
| "epoch": 0.011553052074951144, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005729257129132748, |
| "learning_rate": 1e-05, |
| "loss": -0.0672, |
| "num_tokens": 9026771.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.832338809967041, |
| "sampling/importance_sampling_ratio/mean": 1.0003610849380493, |
| "sampling/importance_sampling_ratio/min": 0.3870709538459778, |
| "sampling/sampling_logp_difference/max": 0.9491472244262695, |
| "sampling/sampling_logp_difference/mean": 0.01959267072379589, |
| "step": 201 |
| }, |
| { |
| "clip_ratio/high_max": 0.00011046524923585821, |
| "clip_ratio/high_mean": 0.00011046524923585821, |
| "clip_ratio/low_mean": 0.00018594362700241618, |
| "clip_ratio/low_min": 0.00018594362700241618, |
| "clip_ratio/region_mean": 0.0002964088762382744, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14093.0, |
| "completions/mean_length": 10794.625, |
| "completions/mean_terminated_length": 9996.1435546875, |
| "completions/min_length": 6482.0, |
| "completions/min_terminated_length": 6482.0, |
| "entropy": 0.6566261723637581, |
| "epoch": 0.011610529945970801, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03386719897389412, |
| "learning_rate": 1e-05, |
| "loss": 0.1077, |
| "num_tokens": 9114312.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997841715812683, |
| "sampling/importance_sampling_ratio/min": 0.36285603046417236, |
| "sampling/sampling_logp_difference/max": 1.013749122619629, |
| "sampling/sampling_logp_difference/mean": 0.02372993715107441, |
| "step": 202 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010559167640167288, |
| "clip_ratio/high_mean": 0.00010559167640167288, |
| "clip_ratio/low_mean": 6.103515625e-05, |
| "clip_ratio/low_min": 6.103515625e-05, |
| "clip_ratio/region_mean": 0.00016662683265167288, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15414.0, |
| "completions/mean_length": 11427.875, |
| "completions/mean_terminated_length": 10719.857421875, |
| "completions/min_length": 8661.0, |
| "completions/min_terminated_length": 8661.0, |
| "entropy": 0.3644193671643734, |
| "epoch": 0.011668007816990459, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.004561516456305981, |
| "learning_rate": 1e-05, |
| "loss": 0.1532, |
| "num_tokens": 9206919.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999821782112122, |
| "sampling/importance_sampling_ratio/min": 0.19193419814109802, |
| "sampling/sampling_logp_difference/max": 1.6506026983261108, |
| "sampling/sampling_logp_difference/mean": 0.01567949913442135, |
| "step": 203 |
| }, |
| { |
| "clip_ratio/high_max": 6.510417006211355e-05, |
| "clip_ratio/high_mean": 6.510417006211355e-05, |
| "clip_ratio/low_mean": 4.36300178989768e-05, |
| "clip_ratio/low_min": 4.36300178989768e-05, |
| "clip_ratio/region_mean": 0.00010873418796109036, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3913.0, |
| "completions/max_terminated_length": 3913.0, |
| "completions/mean_length": 2745.625, |
| "completions/mean_terminated_length": 2745.625, |
| "completions/min_length": 1920.0, |
| "completions/min_terminated_length": 1920.0, |
| "entropy": 0.29109179601073265, |
| "epoch": 0.011725485688010117, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008285376243293285, |
| "learning_rate": 1e-05, |
| "loss": 0.0152, |
| "num_tokens": 9230540.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6389014720916748, |
| "sampling/importance_sampling_ratio/mean": 1.0000051259994507, |
| "sampling/importance_sampling_ratio/min": 0.6845753192901611, |
| "sampling/sampling_logp_difference/max": 0.49402618408203125, |
| "sampling/sampling_logp_difference/mean": 0.012220971286296844, |
| "step": 204 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4695.0, |
| "completions/max_terminated_length": 4695.0, |
| "completions/mean_length": 2600.625, |
| "completions/mean_terminated_length": 2600.625, |
| "completions/min_length": 1549.0, |
| "completions/min_terminated_length": 1549.0, |
| "entropy": 0.34315982833504677, |
| "epoch": 0.011782963559029774, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9252385.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3601510524749756, |
| "sampling/importance_sampling_ratio/mean": 0.9998400211334229, |
| "sampling/importance_sampling_ratio/min": 0.6423700451850891, |
| "sampling/sampling_logp_difference/max": 0.44259071350097656, |
| "sampling/sampling_logp_difference/mean": 0.013423820957541466, |
| "step": 205 |
| }, |
| { |
| "clip_ratio/high_max": 8.21611683932133e-06, |
| "clip_ratio/high_mean": 8.21611683932133e-06, |
| "clip_ratio/low_mean": 0.0006934340599400457, |
| "clip_ratio/low_min": 0.0006934340599400457, |
| "clip_ratio/region_mean": 0.000701650176779367, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15214.0, |
| "completions/mean_length": 13387.5, |
| "completions/mean_terminated_length": 10391.0, |
| "completions/min_length": 4242.0, |
| "completions/min_terminated_length": 4242.0, |
| "entropy": 0.43935373052954674, |
| "epoch": 0.01184044143004943, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005486832931637764, |
| "learning_rate": 1e-05, |
| "loss": -0.0484, |
| "num_tokens": 9361069.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000028133392334, |
| "sampling/importance_sampling_ratio/min": 0.007944649085402489, |
| "sampling/sampling_logp_difference/max": 4.835256576538086, |
| "sampling/sampling_logp_difference/mean": 0.017225248739123344, |
| "step": 206 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1860.0, |
| "completions/max_terminated_length": 1860.0, |
| "completions/mean_length": 1221.25, |
| "completions/mean_terminated_length": 1221.25, |
| "completions/min_length": 875.0, |
| "completions/min_terminated_length": 875.0, |
| "entropy": 0.2764971349388361, |
| "epoch": 0.011897919301069088, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9372015.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2674356698989868, |
| "sampling/importance_sampling_ratio/mean": 0.9998180270195007, |
| "sampling/importance_sampling_ratio/min": 0.6043380498886108, |
| "sampling/sampling_logp_difference/max": 0.5036215782165527, |
| "sampling/sampling_logp_difference/mean": 0.012156876735389233, |
| "step": 207 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16378.0, |
| "completions/mean_length": 12198.0, |
| "completions/mean_terminated_length": 10802.6669921875, |
| "completions/min_length": 5198.0, |
| "completions/min_terminated_length": 5198.0, |
| "entropy": 0.8942966908216476, |
| "epoch": 0.011955397172088746, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9471783.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000622272491455, |
| "sampling/importance_sampling_ratio/min": 0.3456362187862396, |
| "sampling/sampling_logp_difference/max": 1.062368392944336, |
| "sampling/sampling_logp_difference/mean": 0.030848519876599312, |
| "step": 208 |
| }, |
| { |
| "clip_ratio/high_max": 2.7043628506362438e-05, |
| "clip_ratio/high_mean": 2.7043628506362438e-05, |
| "clip_ratio/low_mean": 0.00016821823646751, |
| "clip_ratio/low_min": 0.00016821823646751, |
| "clip_ratio/region_mean": 0.00019526186497387243, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14125.0, |
| "completions/mean_length": 10032.875, |
| "completions/mean_terminated_length": 9125.572265625, |
| "completions/min_length": 3332.0, |
| "completions/min_terminated_length": 3332.0, |
| "entropy": 0.8033781796693802, |
| "epoch": 0.012012875043108403, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011191771365702152, |
| "learning_rate": 1e-05, |
| "loss": 0.2732, |
| "num_tokens": 9553038.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999014735221863, |
| "sampling/importance_sampling_ratio/min": 0.019850613549351692, |
| "sampling/sampling_logp_difference/max": 3.919520378112793, |
| "sampling/sampling_logp_difference/mean": 0.023044193163514137, |
| "step": 209 |
| }, |
| { |
| "clip_ratio/high_max": 3.402286165510304e-05, |
| "clip_ratio/high_mean": 3.402286165510304e-05, |
| "clip_ratio/low_mean": 4.693954178947024e-05, |
| "clip_ratio/low_min": 4.693954178947024e-05, |
| "clip_ratio/region_mean": 8.096240344457328e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3674.0, |
| "completions/max_terminated_length": 3674.0, |
| "completions/mean_length": 2677.625, |
| "completions/mean_terminated_length": 2677.625, |
| "completions/min_length": 1793.0, |
| "completions/min_terminated_length": 1793.0, |
| "entropy": 0.2406750489026308, |
| "epoch": 0.012070352914128061, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009171653538942337, |
| "learning_rate": 1e-05, |
| "loss": -0.0023, |
| "num_tokens": 9576307.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.3651903867721558, |
| "sampling/importance_sampling_ratio/mean": 0.9999209046363831, |
| "sampling/importance_sampling_ratio/min": 0.6137738823890686, |
| "sampling/sampling_logp_difference/max": 0.488128662109375, |
| "sampling/sampling_logp_difference/mean": 0.010427516885101795, |
| "step": 210 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0007341744203586131, |
| "clip_ratio/low_min": 0.0007341744203586131, |
| "clip_ratio/region_mean": 0.0007341744203586131, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3958.0, |
| "completions/max_terminated_length": 3958.0, |
| "completions/mean_length": 1602.75, |
| "completions/mean_terminated_length": 1602.75, |
| "completions/min_length": 652.0, |
| "completions/min_terminated_length": 652.0, |
| "entropy": 0.4816881865262985, |
| "epoch": 0.012127830785147719, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05415642634034157, |
| "learning_rate": 1e-05, |
| "loss": -0.534, |
| "num_tokens": 9590377.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.3418947458267212, |
| "sampling/importance_sampling_ratio/mean": 1.0001122951507568, |
| "sampling/importance_sampling_ratio/min": 0.6219221949577332, |
| "sampling/sampling_logp_difference/max": 0.47494029998779297, |
| "sampling/sampling_logp_difference/mean": 0.021240253001451492, |
| "step": 211 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6170.0, |
| "completions/max_terminated_length": 6170.0, |
| "completions/mean_length": 3812.25, |
| "completions/mean_terminated_length": 3812.25, |
| "completions/min_length": 1433.0, |
| "completions/min_terminated_length": 1433.0, |
| "entropy": 0.5724687688052654, |
| "epoch": 0.012185308656167376, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9621771.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4833757877349854, |
| "sampling/importance_sampling_ratio/mean": 1.0000051259994507, |
| "sampling/importance_sampling_ratio/min": 0.6280277967453003, |
| "sampling/sampling_logp_difference/max": 0.46517086029052734, |
| "sampling/sampling_logp_difference/mean": 0.015964122489094734, |
| "step": 212 |
| }, |
| { |
| "clip_ratio/high_max": 5.3949072025716305e-05, |
| "clip_ratio/high_mean": 5.3949072025716305e-05, |
| "clip_ratio/low_mean": 0.0003672853927128017, |
| "clip_ratio/low_min": 0.0003672853927128017, |
| "clip_ratio/region_mean": 0.000421234464738518, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2317.0, |
| "completions/max_terminated_length": 2317.0, |
| "completions/mean_length": 1286.375, |
| "completions/mean_terminated_length": 1286.375, |
| "completions/min_length": 469.0, |
| "completions/min_terminated_length": 469.0, |
| "entropy": 0.616693627089262, |
| "epoch": 0.012242786527187032, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.11791826784610748, |
| "learning_rate": 1e-05, |
| "loss": -0.5173, |
| "num_tokens": 9632854.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.4110692739486694, |
| "sampling/importance_sampling_ratio/mean": 1.0003750324249268, |
| "sampling/importance_sampling_ratio/min": 0.623399555683136, |
| "sampling/sampling_logp_difference/max": 0.4725675582885742, |
| "sampling/sampling_logp_difference/mean": 0.025106515735387802, |
| "step": 213 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2425.0, |
| "completions/max_terminated_length": 2425.0, |
| "completions/mean_length": 2065.875, |
| "completions/mean_terminated_length": 2065.875, |
| "completions/min_length": 1775.0, |
| "completions/min_terminated_length": 1775.0, |
| "entropy": 0.27775036357343197, |
| "epoch": 0.01230026439820669, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9650789.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3647032976150513, |
| "sampling/importance_sampling_ratio/mean": 0.9997227787971497, |
| "sampling/importance_sampling_ratio/min": 0.6230998635292053, |
| "sampling/sampling_logp_difference/max": 0.47304844856262207, |
| "sampling/sampling_logp_difference/mean": 0.012043694034218788, |
| "step": 214 |
| }, |
| { |
| "clip_ratio/high_max": 7.259000994963571e-05, |
| "clip_ratio/high_mean": 7.259000994963571e-05, |
| "clip_ratio/low_mean": 0.0008480450778733939, |
| "clip_ratio/low_min": 0.0008480450778733939, |
| "clip_ratio/region_mean": 0.0009206350878230296, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14367.0, |
| "completions/mean_length": 11202.75, |
| "completions/mean_terminated_length": 9475.6669921875, |
| "completions/min_length": 6518.0, |
| "completions/min_terminated_length": 6518.0, |
| "entropy": 0.49712123721838, |
| "epoch": 0.012357742269226347, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010083692148327827, |
| "learning_rate": 1e-05, |
| "loss": 0.0807, |
| "num_tokens": 9742363.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002573728561401, |
| "sampling/importance_sampling_ratio/min": 0.1632498800754547, |
| "sampling/sampling_logp_difference/max": 1.8124732971191406, |
| "sampling/sampling_logp_difference/mean": 0.020661983639001846, |
| "step": 215 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14677.0, |
| "completions/mean_length": 12185.375, |
| "completions/mean_terminated_length": 10785.833984375, |
| "completions/min_length": 3584.0, |
| "completions/min_terminated_length": 3584.0, |
| "entropy": 0.6123535353690386, |
| "epoch": 0.012415220140246005, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 9840670.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999887228012085, |
| "sampling/importance_sampling_ratio/min": 0.3155209720134735, |
| "sampling/sampling_logp_difference/max": 1.1535301208496094, |
| "sampling/sampling_logp_difference/mean": 0.025949751958251, |
| "step": 216 |
| }, |
| { |
| "clip_ratio/high_max": 0.00018627778990776278, |
| "clip_ratio/high_mean": 0.00018627778990776278, |
| "clip_ratio/low_mean": 0.000316828147333581, |
| "clip_ratio/low_min": 0.000316828147333581, |
| "clip_ratio/region_mean": 0.0005031059372413438, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11685.0, |
| "completions/max_terminated_length": 11685.0, |
| "completions/mean_length": 6222.0, |
| "completions/mean_terminated_length": 6222.0, |
| "completions/min_length": 1375.0, |
| "completions/min_terminated_length": 1375.0, |
| "entropy": 0.8033790811896324, |
| "epoch": 0.012472698011265663, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013911571353673935, |
| "learning_rate": 1e-05, |
| "loss": 0.2239, |
| "num_tokens": 9891734.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997472167015076, |
| "sampling/importance_sampling_ratio/min": 0.3719465434551239, |
| "sampling/sampling_logp_difference/max": 0.9890050888061523, |
| "sampling/sampling_logp_difference/mean": 0.031158994883298874, |
| "step": 217 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001009012121357955, |
| "clip_ratio/high_mean": 0.0001009012121357955, |
| "clip_ratio/low_mean": 0.00027443770159152336, |
| "clip_ratio/low_min": 0.00027443770159152336, |
| "clip_ratio/region_mean": 0.00037533891372731887, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15150.0, |
| "completions/max_terminated_length": 15150.0, |
| "completions/mean_length": 11302.75, |
| "completions/mean_terminated_length": 11302.75, |
| "completions/min_length": 7471.0, |
| "completions/min_terminated_length": 7471.0, |
| "entropy": 0.4500351585447788, |
| "epoch": 0.01253017588228532, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00955692958086729, |
| "learning_rate": 1e-05, |
| "loss": 0.1302, |
| "num_tokens": 9983764.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999442100524902, |
| "sampling/importance_sampling_ratio/min": 0.005013084504753351, |
| "sampling/sampling_logp_difference/max": 5.295703887939453, |
| "sampling/sampling_logp_difference/mean": 0.019716009497642517, |
| "step": 218 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6111.0, |
| "completions/max_terminated_length": 6111.0, |
| "completions/mean_length": 4018.125, |
| "completions/mean_terminated_length": 4018.125, |
| "completions/min_length": 3109.0, |
| "completions/min_terminated_length": 3109.0, |
| "entropy": 0.5289942175149918, |
| "epoch": 0.012587653753304978, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10017245.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4753402471542358, |
| "sampling/importance_sampling_ratio/mean": 0.9994773268699646, |
| "sampling/importance_sampling_ratio/min": 0.5852012038230896, |
| "sampling/sampling_logp_difference/max": 0.535799503326416, |
| "sampling/sampling_logp_difference/mean": 0.01781732775270939, |
| "step": 219 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010511302025406621, |
| "clip_ratio/high_mean": 0.00010511302025406621, |
| "clip_ratio/low_mean": 0.0004118060605833307, |
| "clip_ratio/low_min": 0.0004118060605833307, |
| "clip_ratio/region_mean": 0.0005169190808373969, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15955.0, |
| "completions/max_terminated_length": 15955.0, |
| "completions/mean_length": 9534.75, |
| "completions/mean_terminated_length": 9534.75, |
| "completions/min_length": 4445.0, |
| "completions/min_terminated_length": 4445.0, |
| "entropy": 0.6315775439143181, |
| "epoch": 0.012645131624324636, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03455370292067528, |
| "learning_rate": 1e-05, |
| "loss": 0.4175, |
| "num_tokens": 10094955.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998697638511658, |
| "sampling/importance_sampling_ratio/min": 0.07237821817398071, |
| "sampling/sampling_logp_difference/max": 2.625849962234497, |
| "sampling/sampling_logp_difference/mean": 0.026303892955183983, |
| "step": 220 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12028.0, |
| "completions/max_terminated_length": 12028.0, |
| "completions/mean_length": 7468.625, |
| "completions/mean_terminated_length": 7468.625, |
| "completions/min_length": 3702.0, |
| "completions/min_terminated_length": 3702.0, |
| "entropy": 0.4179515987634659, |
| "epoch": 0.012702609495344292, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10155752.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7167187929153442, |
| "sampling/importance_sampling_ratio/mean": 1.0000470876693726, |
| "sampling/importance_sampling_ratio/min": 0.38231536746025085, |
| "sampling/sampling_logp_difference/max": 0.9615094661712646, |
| "sampling/sampling_logp_difference/mean": 0.01747564598917961, |
| "step": 221 |
| }, |
| { |
| "clip_ratio/high_max": 3.70562520402018e-05, |
| "clip_ratio/high_mean": 3.70562520402018e-05, |
| "clip_ratio/low_mean": 0.0004943426047248067, |
| "clip_ratio/low_min": 0.0004943426047248067, |
| "clip_ratio/region_mean": 0.0005313988567650085, |
| "completions/clipped_ratio": 0.625, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13493.0, |
| "completions/mean_length": 14093.375, |
| "completions/mean_terminated_length": 10275.6669921875, |
| "completions/min_length": 5608.0, |
| "completions/min_terminated_length": 5608.0, |
| "entropy": 0.40119199454784393, |
| "epoch": 0.01276008736636395, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008288045413792133, |
| "learning_rate": 1e-05, |
| "loss": 0.174, |
| "num_tokens": 10269739.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998534917831421, |
| "sampling/importance_sampling_ratio/min": 0.025376958772540092, |
| "sampling/sampling_logp_difference/max": 3.6739137172698975, |
| "sampling/sampling_logp_difference/mean": 0.018384510651230812, |
| "step": 222 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3782.0, |
| "completions/max_terminated_length": 3782.0, |
| "completions/mean_length": 2814.125, |
| "completions/mean_terminated_length": 2814.125, |
| "completions/min_length": 2102.0, |
| "completions/min_terminated_length": 2102.0, |
| "entropy": 0.2692510113120079, |
| "epoch": 0.012817565237383607, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10292940.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4655894041061401, |
| "sampling/importance_sampling_ratio/mean": 0.9999130368232727, |
| "sampling/importance_sampling_ratio/min": 0.6186597943305969, |
| "sampling/sampling_logp_difference/max": 0.48019981384277344, |
| "sampling/sampling_logp_difference/mean": 0.010910087265074253, |
| "step": 223 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1424.0, |
| "completions/max_terminated_length": 1424.0, |
| "completions/mean_length": 909.25, |
| "completions/mean_terminated_length": 909.25, |
| "completions/min_length": 502.0, |
| "completions/min_terminated_length": 502.0, |
| "entropy": 0.32026074081659317, |
| "epoch": 0.012875043108403265, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10300974.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5538291931152344, |
| "sampling/importance_sampling_ratio/mean": 1.0001140832901, |
| "sampling/importance_sampling_ratio/min": 0.6256819367408752, |
| "sampling/sampling_logp_difference/max": 0.46891307830810547, |
| "sampling/sampling_logp_difference/mean": 0.014822527766227722, |
| "step": 224 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10535.0, |
| "completions/max_terminated_length": 10535.0, |
| "completions/mean_length": 3011.75, |
| "completions/mean_terminated_length": 3011.75, |
| "completions/min_length": 1347.0, |
| "completions/min_terminated_length": 1347.0, |
| "entropy": 0.39564137905836105, |
| "epoch": 0.012932520979422922, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10326124.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5386968851089478, |
| "sampling/importance_sampling_ratio/mean": 1.00032377243042, |
| "sampling/importance_sampling_ratio/min": 0.6109485626220703, |
| "sampling/sampling_logp_difference/max": 0.49274253845214844, |
| "sampling/sampling_logp_difference/mean": 0.019274462014436722, |
| "step": 225 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7352.0, |
| "completions/max_terminated_length": 7352.0, |
| "completions/mean_length": 3362.625, |
| "completions/mean_terminated_length": 3362.625, |
| "completions/min_length": 1670.0, |
| "completions/min_terminated_length": 1670.0, |
| "entropy": 0.4048736020922661, |
| "epoch": 0.01298999885044258, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10353761.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4652092456817627, |
| "sampling/importance_sampling_ratio/mean": 0.9997941255569458, |
| "sampling/importance_sampling_ratio/min": 0.5652009844779968, |
| "sampling/sampling_logp_difference/max": 0.5705738067626953, |
| "sampling/sampling_logp_difference/mean": 0.017642401158809662, |
| "step": 226 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5755.0, |
| "completions/max_terminated_length": 5755.0, |
| "completions/mean_length": 2942.5, |
| "completions/mean_terminated_length": 2942.5, |
| "completions/min_length": 883.0, |
| "completions/min_terminated_length": 883.0, |
| "entropy": 0.8773008808493614, |
| "epoch": 0.013047476721462238, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10378221.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.842469573020935, |
| "sampling/importance_sampling_ratio/mean": 1.000606656074524, |
| "sampling/importance_sampling_ratio/min": 0.616563081741333, |
| "sampling/sampling_logp_difference/max": 0.6111068725585938, |
| "sampling/sampling_logp_difference/mean": 0.025237642228603363, |
| "step": 227 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10473.0, |
| "completions/max_terminated_length": 10473.0, |
| "completions/mean_length": 3174.875, |
| "completions/mean_terminated_length": 3174.875, |
| "completions/min_length": 566.0, |
| "completions/min_terminated_length": 566.0, |
| "entropy": 0.2972034942358732, |
| "epoch": 0.013104954592481894, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10404676.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4848393201828003, |
| "sampling/importance_sampling_ratio/mean": 1.0002814531326294, |
| "sampling/importance_sampling_ratio/min": 0.6210651397705078, |
| "sampling/sampling_logp_difference/max": 0.4763193130493164, |
| "sampling/sampling_logp_difference/mean": 0.012838711962103844, |
| "step": 228 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014502731937682256, |
| "clip_ratio/high_mean": 0.00014502731937682256, |
| "clip_ratio/low_mean": 4.468275074032135e-05, |
| "clip_ratio/low_min": 4.468275074032135e-05, |
| "clip_ratio/region_mean": 0.00018971007011714391, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5595.0, |
| "completions/max_terminated_length": 5595.0, |
| "completions/mean_length": 4151.0, |
| "completions/mean_terminated_length": 4151.0, |
| "completions/min_length": 3123.0, |
| "completions/min_terminated_length": 3123.0, |
| "entropy": 0.5627840459346771, |
| "epoch": 0.013162432463501551, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011595550924539566, |
| "learning_rate": 1e-05, |
| "loss": 0.1234, |
| "num_tokens": 10440100.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.7026751041412354, |
| "sampling/importance_sampling_ratio/mean": 1.0003058910369873, |
| "sampling/importance_sampling_ratio/min": 0.5480839014053345, |
| "sampling/sampling_logp_difference/max": 0.6013269424438477, |
| "sampling/sampling_logp_difference/mean": 0.020583635196089745, |
| "step": 229 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010347706665925216, |
| "clip_ratio/high_mean": 0.00010347706665925216, |
| "clip_ratio/low_mean": 8.392333984375e-05, |
| "clip_ratio/low_min": 8.392333984375e-05, |
| "clip_ratio/region_mean": 0.00018740040650300216, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12809.0, |
| "completions/mean_length": 10427.125, |
| "completions/mean_terminated_length": 9576.1435546875, |
| "completions/min_length": 6924.0, |
| "completions/min_terminated_length": 6924.0, |
| "entropy": 0.4362257868051529, |
| "epoch": 0.013219910334521209, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005803901236504316, |
| "learning_rate": 1e-05, |
| "loss": 0.202, |
| "num_tokens": 10525341.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000170111656189, |
| "sampling/importance_sampling_ratio/min": 0.26937708258628845, |
| "sampling/sampling_logp_difference/max": 1.311643123626709, |
| "sampling/sampling_logp_difference/mean": 0.01951228268444538, |
| "step": 230 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3820.0, |
| "completions/max_terminated_length": 3820.0, |
| "completions/mean_length": 2411.375, |
| "completions/mean_terminated_length": 2411.375, |
| "completions/min_length": 1320.0, |
| "completions/min_terminated_length": 1320.0, |
| "entropy": 0.3224306609481573, |
| "epoch": 0.013277388205540867, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10545624.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5646144151687622, |
| "sampling/importance_sampling_ratio/mean": 0.9997329115867615, |
| "sampling/importance_sampling_ratio/min": 0.6744040250778198, |
| "sampling/sampling_logp_difference/max": 0.44763946533203125, |
| "sampling/sampling_logp_difference/mean": 0.013830232433974743, |
| "step": 231 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0005034416935814079, |
| "clip_ratio/low_min": 0.0005034416935814079, |
| "clip_ratio/region_mean": 0.0005034416935814079, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8818.0, |
| "completions/max_terminated_length": 8818.0, |
| "completions/mean_length": 5082.75, |
| "completions/mean_terminated_length": 5082.75, |
| "completions/min_length": 1564.0, |
| "completions/min_terminated_length": 1564.0, |
| "entropy": 0.5696996338665485, |
| "epoch": 0.013334866076560524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006232178304344416, |
| "learning_rate": 1e-05, |
| "loss": 0.028, |
| "num_tokens": 10587430.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.537739634513855, |
| "sampling/importance_sampling_ratio/mean": 1.0001429319381714, |
| "sampling/importance_sampling_ratio/min": 0.4392375946044922, |
| "sampling/sampling_logp_difference/max": 0.8227148056030273, |
| "sampling/sampling_logp_difference/mean": 0.02268536016345024, |
| "step": 232 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7178.0, |
| "completions/max_terminated_length": 7178.0, |
| "completions/mean_length": 3350.5, |
| "completions/mean_terminated_length": 3350.5, |
| "completions/min_length": 1589.0, |
| "completions/min_terminated_length": 1589.0, |
| "entropy": 0.41542892530560493, |
| "epoch": 0.013392343947580182, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10615994.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5274951457977295, |
| "sampling/importance_sampling_ratio/mean": 1.0001320838928223, |
| "sampling/importance_sampling_ratio/min": 0.5196384191513062, |
| "sampling/sampling_logp_difference/max": 0.6546220779418945, |
| "sampling/sampling_logp_difference/mean": 0.01948987878859043, |
| "step": 233 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9857.0, |
| "completions/max_terminated_length": 9857.0, |
| "completions/mean_length": 5487.5, |
| "completions/mean_terminated_length": 5487.5, |
| "completions/min_length": 3320.0, |
| "completions/min_terminated_length": 3320.0, |
| "entropy": 0.38110455498099327, |
| "epoch": 0.01344982181859984, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10660734.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.81279456615448, |
| "sampling/importance_sampling_ratio/mean": 1.00006103515625, |
| "sampling/importance_sampling_ratio/min": 0.581904947757721, |
| "sampling/sampling_logp_difference/max": 0.5948696136474609, |
| "sampling/sampling_logp_difference/mean": 0.015486331656575203, |
| "step": 234 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12979.0, |
| "completions/max_terminated_length": 12979.0, |
| "completions/mean_length": 5788.625, |
| "completions/mean_terminated_length": 5788.625, |
| "completions/min_length": 2136.0, |
| "completions/min_terminated_length": 2136.0, |
| "entropy": 0.4809899814426899, |
| "epoch": 0.013507299689619497, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10708523.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5933127403259277, |
| "sampling/importance_sampling_ratio/mean": 0.9999656677246094, |
| "sampling/importance_sampling_ratio/min": 0.2556094825267792, |
| "sampling/sampling_logp_difference/max": 1.3641045093536377, |
| "sampling/sampling_logp_difference/mean": 0.017151974141597748, |
| "step": 235 |
| }, |
| { |
| "clip_ratio/high_max": 2.4659042537678033e-05, |
| "clip_ratio/high_mean": 2.4659042537678033e-05, |
| "clip_ratio/low_mean": 0.0005239465244812891, |
| "clip_ratio/low_min": 0.0005239465244812891, |
| "clip_ratio/region_mean": 0.0005486055670189671, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14968.0, |
| "completions/mean_length": 12166.25, |
| "completions/mean_terminated_length": 7948.5, |
| "completions/min_length": 4181.0, |
| "completions/min_terminated_length": 4181.0, |
| "entropy": 0.4801964499056339, |
| "epoch": 0.013564777560639153, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005800375249236822, |
| "learning_rate": 1e-05, |
| "loss": 0.1765, |
| "num_tokens": 10807061.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998948574066162, |
| "sampling/importance_sampling_ratio/min": 6.112750128295374e-08, |
| "sampling/sampling_logp_difference/max": 16.61030387878418, |
| "sampling/sampling_logp_difference/mean": 0.020769845694303513, |
| "step": 236 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16148.0, |
| "completions/mean_length": 13540.125, |
| "completions/mean_terminated_length": 12592.1669921875, |
| "completions/min_length": 9726.0, |
| "completions/min_terminated_length": 9726.0, |
| "entropy": 0.9904661029577255, |
| "epoch": 0.013622255431658811, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10916926.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997573494911194, |
| "sampling/importance_sampling_ratio/min": 8.059122774284333e-05, |
| "sampling/sampling_logp_difference/max": 9.42612075805664, |
| "sampling/sampling_logp_difference/mean": 0.032993484288454056, |
| "step": 237 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4654.0, |
| "completions/max_terminated_length": 4654.0, |
| "completions/mean_length": 2031.25, |
| "completions/mean_terminated_length": 2031.25, |
| "completions/min_length": 983.0, |
| "completions/min_terminated_length": 983.0, |
| "entropy": 0.23056718334555626, |
| "epoch": 0.013679733302678469, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 10934704.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5085855722427368, |
| "sampling/importance_sampling_ratio/mean": 0.9995875358581543, |
| "sampling/importance_sampling_ratio/min": 0.6115107536315918, |
| "sampling/sampling_logp_difference/max": 0.4918227195739746, |
| "sampling/sampling_logp_difference/mean": 0.01120684016495943, |
| "step": 238 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 7.768800423946232e-05, |
| "clip_ratio/low_min": 7.768800423946232e-05, |
| "clip_ratio/region_mean": 7.768800423946232e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6228.0, |
| "completions/max_terminated_length": 6228.0, |
| "completions/mean_length": 2806.25, |
| "completions/mean_terminated_length": 2806.25, |
| "completions/min_length": 998.0, |
| "completions/min_terminated_length": 998.0, |
| "entropy": 1.0021002143621445, |
| "epoch": 0.013737211173698126, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009687749668955803, |
| "learning_rate": 1e-05, |
| "loss": -0.1695, |
| "num_tokens": 10958122.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4409617185592651, |
| "sampling/importance_sampling_ratio/mean": 0.9998266100883484, |
| "sampling/importance_sampling_ratio/min": 0.2617310881614685, |
| "sampling/sampling_logp_difference/max": 1.340437650680542, |
| "sampling/sampling_logp_difference/mean": 0.024621441960334778, |
| "step": 239 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12936.0, |
| "completions/max_terminated_length": 12936.0, |
| "completions/mean_length": 8497.25, |
| "completions/mean_terminated_length": 8497.25, |
| "completions/min_length": 4639.0, |
| "completions/min_terminated_length": 4639.0, |
| "entropy": 0.5409430265426636, |
| "epoch": 0.013794689044717784, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11026940.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9139245748519897, |
| "sampling/importance_sampling_ratio/mean": 0.9997201561927795, |
| "sampling/importance_sampling_ratio/min": 0.35531097650527954, |
| "sampling/sampling_logp_difference/max": 1.034761905670166, |
| "sampling/sampling_logp_difference/mean": 0.020615391433238983, |
| "step": 240 |
| }, |
| { |
| "clip_ratio/high_max": 2.386103369644843e-05, |
| "clip_ratio/high_mean": 2.386103369644843e-05, |
| "clip_ratio/low_mean": 0.0006688979556201957, |
| "clip_ratio/low_min": 0.0006688979556201957, |
| "clip_ratio/region_mean": 0.0006927589893166441, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15716.0, |
| "completions/mean_length": 11776.375, |
| "completions/mean_terminated_length": 11118.1435546875, |
| "completions/min_length": 7469.0, |
| "completions/min_terminated_length": 7469.0, |
| "entropy": 0.6553877890110016, |
| "epoch": 0.013852166915737442, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005486266687512398, |
| "learning_rate": 1e-05, |
| "loss": -0.1183, |
| "num_tokens": 11122135.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000144243240356, |
| "sampling/importance_sampling_ratio/min": 0.010656909085810184, |
| "sampling/sampling_logp_difference/max": 4.541546821594238, |
| "sampling/sampling_logp_difference/mean": 0.024141818284988403, |
| "step": 241 |
| }, |
| { |
| "clip_ratio/high_max": 1.9644821804831736e-05, |
| "clip_ratio/high_mean": 1.9644821804831736e-05, |
| "clip_ratio/low_mean": 0.00018991712477145484, |
| "clip_ratio/low_min": 0.00018991712477145484, |
| "clip_ratio/region_mean": 0.00020956194657628657, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10198.0, |
| "completions/max_terminated_length": 10198.0, |
| "completions/mean_length": 8131.625, |
| "completions/mean_terminated_length": 8131.625, |
| "completions/min_length": 6354.0, |
| "completions/min_terminated_length": 6354.0, |
| "entropy": 0.890590064227581, |
| "epoch": 0.0139096447867571, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.013450238853693008, |
| "learning_rate": 1e-05, |
| "loss": -0.0102, |
| "num_tokens": 11188132.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.781151533126831, |
| "sampling/importance_sampling_ratio/mean": 0.9999437928199768, |
| "sampling/importance_sampling_ratio/min": 0.45966771245002747, |
| "sampling/sampling_logp_difference/max": 0.7772514820098877, |
| "sampling/sampling_logp_difference/mean": 0.02461942844092846, |
| "step": 242 |
| }, |
| { |
| "clip_ratio/high_max": 6.480041338363662e-05, |
| "clip_ratio/high_mean": 6.480041338363662e-05, |
| "clip_ratio/low_mean": 6.854949606349692e-05, |
| "clip_ratio/low_min": 6.854949606349692e-05, |
| "clip_ratio/region_mean": 0.00013334990944713354, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3913.0, |
| "completions/max_terminated_length": 3913.0, |
| "completions/mean_length": 1979.125, |
| "completions/mean_terminated_length": 1979.125, |
| "completions/min_length": 748.0, |
| "completions/min_terminated_length": 748.0, |
| "entropy": 0.3730858415365219, |
| "epoch": 0.013967122657776755, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07335606217384338, |
| "learning_rate": 1e-05, |
| "loss": 0.0606, |
| "num_tokens": 11205037.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.7882685661315918, |
| "sampling/importance_sampling_ratio/mean": 0.9999749660491943, |
| "sampling/importance_sampling_ratio/min": 0.5497262477874756, |
| "sampling/sampling_logp_difference/max": 0.598334789276123, |
| "sampling/sampling_logp_difference/mean": 0.016838887706398964, |
| "step": 243 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1756.0, |
| "completions/max_terminated_length": 1756.0, |
| "completions/mean_length": 1302.75, |
| "completions/mean_terminated_length": 1302.75, |
| "completions/min_length": 857.0, |
| "completions/min_terminated_length": 857.0, |
| "entropy": 0.23300896771252155, |
| "epoch": 0.014024600528796413, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11216075.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5093194246292114, |
| "sampling/importance_sampling_ratio/mean": 1.0003427267074585, |
| "sampling/importance_sampling_ratio/min": 0.6418024301528931, |
| "sampling/sampling_logp_difference/max": 0.44347476959228516, |
| "sampling/sampling_logp_difference/mean": 0.009614585898816586, |
| "step": 244 |
| }, |
| { |
| "clip_ratio/high_max": 2.7015345040126704e-05, |
| "clip_ratio/high_mean": 2.7015345040126704e-05, |
| "clip_ratio/low_mean": 0.00046870691221556626, |
| "clip_ratio/low_min": 0.00046870691221556626, |
| "clip_ratio/region_mean": 0.000495722257255693, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13101.0, |
| "completions/mean_length": 9661.0, |
| "completions/mean_terminated_length": 8700.572265625, |
| "completions/min_length": 4408.0, |
| "completions/min_terminated_length": 4408.0, |
| "entropy": 0.30384486727416515, |
| "epoch": 0.01408207839981607, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008967713452875614, |
| "learning_rate": 1e-05, |
| "loss": 0.1583, |
| "num_tokens": 11294371.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998798966407776, |
| "sampling/importance_sampling_ratio/min": 0.1724100410938263, |
| "sampling/sampling_logp_difference/max": 1.7578797340393066, |
| "sampling/sampling_logp_difference/mean": 0.014898917637765408, |
| "step": 245 |
| }, |
| { |
| "clip_ratio/high_max": 0.00010216360897175036, |
| "clip_ratio/high_mean": 0.00010216360897175036, |
| "clip_ratio/low_mean": 0.0009872470836853608, |
| "clip_ratio/low_min": 0.0009872470836853608, |
| "clip_ratio/region_mean": 0.0010894106926571112, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13869.0, |
| "completions/mean_length": 11561.375, |
| "completions/mean_terminated_length": 9953.833984375, |
| "completions/min_length": 3559.0, |
| "completions/min_terminated_length": 3559.0, |
| "entropy": 0.6489362716674805, |
| "epoch": 0.014139556270835728, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009956092573702335, |
| "learning_rate": 1e-05, |
| "loss": -0.0269, |
| "num_tokens": 11387998.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000132441520691, |
| "sampling/importance_sampling_ratio/min": 0.02731258235871792, |
| "sampling/sampling_logp_difference/max": 3.600407838821411, |
| "sampling/sampling_logp_difference/mean": 0.02649668976664543, |
| "step": 246 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7619.0, |
| "completions/max_terminated_length": 7619.0, |
| "completions/mean_length": 3206.375, |
| "completions/mean_terminated_length": 3206.375, |
| "completions/min_length": 667.0, |
| "completions/min_terminated_length": 667.0, |
| "entropy": 1.464602880179882, |
| "epoch": 0.014197034141855386, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11415401.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4338622093200684, |
| "sampling/importance_sampling_ratio/mean": 0.9999920725822449, |
| "sampling/importance_sampling_ratio/min": 0.44867074489593506, |
| "sampling/sampling_logp_difference/max": 0.8014659881591797, |
| "sampling/sampling_logp_difference/mean": 0.03130270913243294, |
| "step": 247 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6551.0, |
| "completions/max_terminated_length": 6551.0, |
| "completions/mean_length": 4104.125, |
| "completions/mean_terminated_length": 4104.125, |
| "completions/min_length": 1975.0, |
| "completions/min_terminated_length": 1975.0, |
| "entropy": 0.3689926713705063, |
| "epoch": 0.014254512012875043, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11449394.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5026180744171143, |
| "sampling/importance_sampling_ratio/mean": 0.9996576309204102, |
| "sampling/importance_sampling_ratio/min": 0.5714125633239746, |
| "sampling/sampling_logp_difference/max": 0.5596437454223633, |
| "sampling/sampling_logp_difference/mean": 0.01587335765361786, |
| "step": 248 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15847.0, |
| "completions/max_terminated_length": 15847.0, |
| "completions/mean_length": 11492.625, |
| "completions/mean_terminated_length": 11492.625, |
| "completions/min_length": 8562.0, |
| "completions/min_terminated_length": 8562.0, |
| "entropy": 0.5737185478210449, |
| "epoch": 0.014311989883894701, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11542447.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000476837158203, |
| "sampling/importance_sampling_ratio/min": 0.14115479588508606, |
| "sampling/sampling_logp_difference/max": 1.9578981399536133, |
| "sampling/sampling_logp_difference/mean": 0.02276143990457058, |
| "step": 249 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013715856039198115, |
| "clip_ratio/high_mean": 0.00013715856039198115, |
| "clip_ratio/low_mean": 0.0001709281132207252, |
| "clip_ratio/low_min": 0.0001709281132207252, |
| "clip_ratio/region_mean": 0.00030808667361270636, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7131.0, |
| "completions/max_terminated_length": 7131.0, |
| "completions/mean_length": 4421.625, |
| "completions/mean_terminated_length": 4421.625, |
| "completions/min_length": 1999.0, |
| "completions/min_terminated_length": 1999.0, |
| "entropy": 0.2619555275887251, |
| "epoch": 0.014369467754914359, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011488448828458786, |
| "learning_rate": 1e-05, |
| "loss": 0.2443, |
| "num_tokens": 11580052.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.7280057668685913, |
| "sampling/importance_sampling_ratio/mean": 1.0001354217529297, |
| "sampling/importance_sampling_ratio/min": 0.6068691611289978, |
| "sampling/sampling_logp_difference/max": 0.5469679832458496, |
| "sampling/sampling_logp_difference/mean": 0.012207957915961742, |
| "step": 250 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2934.0, |
| "completions/max_terminated_length": 2934.0, |
| "completions/mean_length": 1419.75, |
| "completions/mean_terminated_length": 1419.75, |
| "completions/min_length": 633.0, |
| "completions/min_terminated_length": 633.0, |
| "entropy": 0.28696313686668873, |
| "epoch": 0.014426945625934015, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11592610.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.443624496459961, |
| "sampling/importance_sampling_ratio/mean": 0.9992883801460266, |
| "sampling/importance_sampling_ratio/min": 0.6237903237342834, |
| "sampling/sampling_logp_difference/max": 0.4719409942626953, |
| "sampling/sampling_logp_difference/mean": 0.01366155780851841, |
| "step": 251 |
| }, |
| { |
| "clip_ratio/high_max": 0.00017145523452199996, |
| "clip_ratio/high_mean": 0.00017145523452199996, |
| "clip_ratio/low_mean": 0.00010699764243327081, |
| "clip_ratio/low_min": 0.00010699764243327081, |
| "clip_ratio/region_mean": 0.00027845287695527077, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13382.0, |
| "completions/max_terminated_length": 13382.0, |
| "completions/mean_length": 8514.75, |
| "completions/mean_terminated_length": 8514.75, |
| "completions/min_length": 5461.0, |
| "completions/min_terminated_length": 5461.0, |
| "entropy": 0.5025256536900997, |
| "epoch": 0.014484423496953672, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005746009759604931, |
| "learning_rate": 1e-05, |
| "loss": 0.0348, |
| "num_tokens": 11661440.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000219702720642, |
| "sampling/importance_sampling_ratio/min": 0.5012786388397217, |
| "sampling/sampling_logp_difference/max": 0.9241889715194702, |
| "sampling/sampling_logp_difference/mean": 0.020174650475382805, |
| "step": 252 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 6.6195942054037e-05, |
| "clip_ratio/low_min": 6.6195942054037e-05, |
| "clip_ratio/region_mean": 6.6195942054037e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5665.0, |
| "completions/max_terminated_length": 5665.0, |
| "completions/mean_length": 3291.5, |
| "completions/mean_terminated_length": 3291.5, |
| "completions/min_length": 1059.0, |
| "completions/min_terminated_length": 1059.0, |
| "entropy": 0.43801668658852577, |
| "epoch": 0.01454190136797333, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005613123066723347, |
| "learning_rate": 1e-05, |
| "loss": 0.2547, |
| "num_tokens": 11688764.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.419911503791809, |
| "sampling/importance_sampling_ratio/mean": 1.0000346899032593, |
| "sampling/importance_sampling_ratio/min": 0.6072993278503418, |
| "sampling/sampling_logp_difference/max": 0.4987335205078125, |
| "sampling/sampling_logp_difference/mean": 0.014720753766596317, |
| "step": 253 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2179.0, |
| "completions/max_terminated_length": 2179.0, |
| "completions/mean_length": 884.375, |
| "completions/mean_terminated_length": 884.375, |
| "completions/min_length": 461.0, |
| "completions/min_terminated_length": 461.0, |
| "entropy": 0.28187943436205387, |
| "epoch": 0.014599379238992988, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11696519.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.696502923965454, |
| "sampling/importance_sampling_ratio/mean": 0.999819278717041, |
| "sampling/importance_sampling_ratio/min": 0.5635024309158325, |
| "sampling/sampling_logp_difference/max": 0.5735836029052734, |
| "sampling/sampling_logp_difference/mean": 0.01351790502667427, |
| "step": 254 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3781.0, |
| "completions/max_terminated_length": 3781.0, |
| "completions/mean_length": 2290.625, |
| "completions/mean_terminated_length": 2290.625, |
| "completions/min_length": 843.0, |
| "completions/min_terminated_length": 843.0, |
| "entropy": 0.3055574521422386, |
| "epoch": 0.014656857110012645, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11716028.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3616334199905396, |
| "sampling/importance_sampling_ratio/mean": 1.0000905990600586, |
| "sampling/importance_sampling_ratio/min": 0.5984509587287903, |
| "sampling/sampling_logp_difference/max": 0.513410747051239, |
| "sampling/sampling_logp_difference/mean": 0.013125410303473473, |
| "step": 255 |
| }, |
| { |
| "clip_ratio/high_max": 5.335124114935752e-05, |
| "clip_ratio/high_mean": 5.335124114935752e-05, |
| "clip_ratio/low_mean": 0.00042540972390270326, |
| "clip_ratio/low_min": 0.00042540972390270326, |
| "clip_ratio/region_mean": 0.0004787609650520608, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15349.0, |
| "completions/mean_length": 12992.0, |
| "completions/mean_terminated_length": 12507.4287109375, |
| "completions/min_length": 8727.0, |
| "completions/min_terminated_length": 8727.0, |
| "entropy": 0.6972725093364716, |
| "epoch": 0.014714334981032303, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009135198779404163, |
| "learning_rate": 1e-05, |
| "loss": 0.0787, |
| "num_tokens": 11820764.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000097393989563, |
| "sampling/importance_sampling_ratio/min": 0.16634340584278107, |
| "sampling/sampling_logp_difference/max": 1.793700933456421, |
| "sampling/sampling_logp_difference/mean": 0.02293052151799202, |
| "step": 256 |
| }, |
| { |
| "clip_ratio/high_max": 4.2088422560482286e-05, |
| "clip_ratio/high_mean": 4.2088422560482286e-05, |
| "clip_ratio/low_mean": 8.430281741311774e-05, |
| "clip_ratio/low_min": 8.430281741311774e-05, |
| "clip_ratio/region_mean": 0.00012639123997360002, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10320.0, |
| "completions/max_terminated_length": 10320.0, |
| "completions/mean_length": 5065.625, |
| "completions/mean_terminated_length": 5065.625, |
| "completions/min_length": 3496.0, |
| "completions/min_terminated_length": 3496.0, |
| "entropy": 0.26933291368186474, |
| "epoch": 0.01477181285205196, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05237021669745445, |
| "learning_rate": 1e-05, |
| "loss": 0.0603, |
| "num_tokens": 11862289.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.5922070741653442, |
| "sampling/importance_sampling_ratio/mean": 0.9999058246612549, |
| "sampling/importance_sampling_ratio/min": 0.5469368100166321, |
| "sampling/sampling_logp_difference/max": 0.6034219861030579, |
| "sampling/sampling_logp_difference/mean": 0.012624706141650677, |
| "step": 257 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012257248636160512, |
| "clip_ratio/high_mean": 0.00012257248636160512, |
| "clip_ratio/low_mean": 0.0001102778987842612, |
| "clip_ratio/low_min": 0.0001102778987842612, |
| "clip_ratio/region_mean": 0.0002328503851458663, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6737.0, |
| "completions/max_terminated_length": 6737.0, |
| "completions/mean_length": 4876.625, |
| "completions/mean_terminated_length": 4876.625, |
| "completions/min_length": 3497.0, |
| "completions/min_terminated_length": 3497.0, |
| "entropy": 0.345124926418066, |
| "epoch": 0.014829290723071617, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05227786675095558, |
| "learning_rate": 1e-05, |
| "loss": -0.0249, |
| "num_tokens": 11904054.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001052618026733, |
| "sampling/importance_sampling_ratio/min": 0.5758959054946899, |
| "sampling/sampling_logp_difference/max": 0.7921288013458252, |
| "sampling/sampling_logp_difference/mean": 0.014587976969778538, |
| "step": 258 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3329.0, |
| "completions/max_terminated_length": 3329.0, |
| "completions/mean_length": 2546.125, |
| "completions/mean_terminated_length": 2546.125, |
| "completions/min_length": 1418.0, |
| "completions/min_terminated_length": 1418.0, |
| "entropy": 0.25126997753977776, |
| "epoch": 0.014886768594091274, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 11925727.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4259095191955566, |
| "sampling/importance_sampling_ratio/mean": 1.0000795125961304, |
| "sampling/importance_sampling_ratio/min": 0.6376045346260071, |
| "sampling/sampling_logp_difference/max": 0.45003700256347656, |
| "sampling/sampling_logp_difference/mean": 0.01043980848044157, |
| "step": 259 |
| }, |
| { |
| "clip_ratio/high_max": 2.7790129024651833e-05, |
| "clip_ratio/high_mean": 2.7790129024651833e-05, |
| "clip_ratio/low_mean": 0.0009038363714353181, |
| "clip_ratio/low_min": 0.0009038363714353181, |
| "clip_ratio/region_mean": 0.00093162650045997, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15221.0, |
| "completions/max_terminated_length": 15221.0, |
| "completions/mean_length": 10797.625, |
| "completions/mean_terminated_length": 10797.625, |
| "completions/min_length": 6488.0, |
| "completions/min_terminated_length": 6488.0, |
| "entropy": 0.872970461845398, |
| "epoch": 0.014944246465110932, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006912244018167257, |
| "learning_rate": 1e-05, |
| "loss": 0.0593, |
| "num_tokens": 12013652.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002055168151855, |
| "sampling/importance_sampling_ratio/min": 0.05752217769622803, |
| "sampling/sampling_logp_difference/max": 2.8555846214294434, |
| "sampling/sampling_logp_difference/mean": 0.03274575620889664, |
| "step": 260 |
| }, |
| { |
| "clip_ratio/high_max": 2.3955539290909655e-05, |
| "clip_ratio/high_mean": 2.3955539290909655e-05, |
| "clip_ratio/low_mean": 6.0837119235657156e-05, |
| "clip_ratio/low_min": 6.0837119235657156e-05, |
| "clip_ratio/region_mean": 8.479265852656681e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10804.0, |
| "completions/max_terminated_length": 10804.0, |
| "completions/mean_length": 6303.625, |
| "completions/mean_terminated_length": 6303.625, |
| "completions/min_length": 4371.0, |
| "completions/min_terminated_length": 4371.0, |
| "entropy": 1.3393941968679428, |
| "epoch": 0.01500172433613059, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0125952810049057, |
| "learning_rate": 1e-05, |
| "loss": -0.071, |
| "num_tokens": 12064993.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4859604835510254, |
| "sampling/importance_sampling_ratio/mean": 0.9997767806053162, |
| "sampling/importance_sampling_ratio/min": 0.3659899830818176, |
| "sampling/sampling_logp_difference/max": 1.0051493644714355, |
| "sampling/sampling_logp_difference/mean": 0.028279611840844154, |
| "step": 261 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14001.0, |
| "completions/max_terminated_length": 14001.0, |
| "completions/mean_length": 9267.75, |
| "completions/mean_terminated_length": 9267.75, |
| "completions/min_length": 4333.0, |
| "completions/min_terminated_length": 4333.0, |
| "entropy": 0.7848407402634621, |
| "epoch": 0.015059202207150247, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12139951.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998934864997864, |
| "sampling/importance_sampling_ratio/min": 0.31592896580696106, |
| "sampling/sampling_logp_difference/max": 1.486649990081787, |
| "sampling/sampling_logp_difference/mean": 0.020874252542853355, |
| "step": 262 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8796.0, |
| "completions/max_terminated_length": 8796.0, |
| "completions/mean_length": 5854.0, |
| "completions/mean_terminated_length": 5854.0, |
| "completions/min_length": 3834.0, |
| "completions/min_terminated_length": 3834.0, |
| "entropy": 0.6252594366669655, |
| "epoch": 0.015116680078169905, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12187543.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9118043184280396, |
| "sampling/importance_sampling_ratio/mean": 0.9999527335166931, |
| "sampling/importance_sampling_ratio/min": 0.6088029742240906, |
| "sampling/sampling_logp_difference/max": 0.6480474472045898, |
| "sampling/sampling_logp_difference/mean": 0.022176310420036316, |
| "step": 263 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3143.0, |
| "completions/max_terminated_length": 3143.0, |
| "completions/mean_length": 1706.125, |
| "completions/mean_terminated_length": 1706.125, |
| "completions/min_length": 677.0, |
| "completions/min_terminated_length": 677.0, |
| "entropy": 0.3878425918519497, |
| "epoch": 0.015174157949189563, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12201816.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.500133752822876, |
| "sampling/importance_sampling_ratio/mean": 1.0005122423171997, |
| "sampling/importance_sampling_ratio/min": 0.6614216566085815, |
| "sampling/sampling_logp_difference/max": 0.4133636951446533, |
| "sampling/sampling_logp_difference/mean": 0.016923129558563232, |
| "step": 264 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10087.0, |
| "completions/max_terminated_length": 10087.0, |
| "completions/mean_length": 5456.75, |
| "completions/mean_terminated_length": 5456.75, |
| "completions/min_length": 3458.0, |
| "completions/min_terminated_length": 3458.0, |
| "entropy": 0.4956372305750847, |
| "epoch": 0.01523163582020922, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12246822.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7300132513046265, |
| "sampling/importance_sampling_ratio/mean": 1.0002621412277222, |
| "sampling/importance_sampling_ratio/min": 0.6069631576538086, |
| "sampling/sampling_logp_difference/max": 0.5481290817260742, |
| "sampling/sampling_logp_difference/mean": 0.01882501319050789, |
| "step": 265 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9249.0, |
| "completions/max_terminated_length": 9249.0, |
| "completions/mean_length": 5672.125, |
| "completions/mean_terminated_length": 5672.125, |
| "completions/min_length": 4063.0, |
| "completions/min_terminated_length": 4063.0, |
| "entropy": 0.38724199309945107, |
| "epoch": 0.015289113691228876, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12293935.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7353522777557373, |
| "sampling/importance_sampling_ratio/mean": 0.9997246265411377, |
| "sampling/importance_sampling_ratio/min": 0.4818136394023895, |
| "sampling/sampling_logp_difference/max": 0.7301979064941406, |
| "sampling/sampling_logp_difference/mean": 0.01586085744202137, |
| "step": 266 |
| }, |
| { |
| "clip_ratio/high_max": 5.242153383733239e-05, |
| "clip_ratio/high_mean": 5.242153383733239e-05, |
| "clip_ratio/low_mean": 0.00019502839313645381, |
| "clip_ratio/low_min": 0.00019502839313645381, |
| "clip_ratio/region_mean": 0.0002474499269737862, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7770.0, |
| "completions/max_terminated_length": 7770.0, |
| "completions/mean_length": 5646.875, |
| "completions/mean_terminated_length": 5646.875, |
| "completions/min_length": 4140.0, |
| "completions/min_terminated_length": 4140.0, |
| "entropy": 0.7165233716368675, |
| "epoch": 0.015346591562248534, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021525582298636436, |
| "learning_rate": 1e-05, |
| "loss": -0.1338, |
| "num_tokens": 12340294.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999697804450989, |
| "sampling/importance_sampling_ratio/min": 0.38163527846336365, |
| "sampling/sampling_logp_difference/max": 1.1706171035766602, |
| "sampling/sampling_logp_difference/mean": 0.01964937150478363, |
| "step": 267 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6600.0, |
| "completions/max_terminated_length": 6600.0, |
| "completions/mean_length": 4055.5, |
| "completions/mean_terminated_length": 4055.5, |
| "completions/min_length": 1793.0, |
| "completions/min_terminated_length": 1793.0, |
| "entropy": 0.49006812646985054, |
| "epoch": 0.015404069433268192, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12374034.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4653664827346802, |
| "sampling/importance_sampling_ratio/mean": 0.9998965263366699, |
| "sampling/importance_sampling_ratio/min": 0.5483196377754211, |
| "sampling/sampling_logp_difference/max": 0.6008968353271484, |
| "sampling/sampling_logp_difference/mean": 0.019602404907345772, |
| "step": 268 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7732.0, |
| "completions/max_terminated_length": 7732.0, |
| "completions/mean_length": 4636.875, |
| "completions/mean_terminated_length": 4636.875, |
| "completions/min_length": 1654.0, |
| "completions/min_terminated_length": 1654.0, |
| "entropy": 0.5612859316170216, |
| "epoch": 0.01546154730428785, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12412681.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.99969083070755, |
| "sampling/importance_sampling_ratio/min": 0.3210018277168274, |
| "sampling/sampling_logp_difference/max": 1.1363084316253662, |
| "sampling/sampling_logp_difference/mean": 0.016598373651504517, |
| "step": 269 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11061.0, |
| "completions/max_terminated_length": 11061.0, |
| "completions/mean_length": 4946.125, |
| "completions/mean_terminated_length": 4946.125, |
| "completions/min_length": 1497.0, |
| "completions/min_terminated_length": 1497.0, |
| "entropy": 0.8044924661517143, |
| "epoch": 0.015519025175307507, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12453498.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999637007713318, |
| "sampling/importance_sampling_ratio/min": 2.09754239222093e-06, |
| "sampling/sampling_logp_difference/max": 13.07474422454834, |
| "sampling/sampling_logp_difference/mean": 0.022526336833834648, |
| "step": 270 |
| }, |
| { |
| "clip_ratio/high_max": 1.0969723916787189e-05, |
| "clip_ratio/high_mean": 1.0969723916787189e-05, |
| "clip_ratio/low_mean": 0.0007923787270556204, |
| "clip_ratio/low_min": 0.0007923787270556204, |
| "clip_ratio/region_mean": 0.0008033484509724076, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15828.0, |
| "completions/mean_length": 14050.25, |
| "completions/mean_terminated_length": 13272.333984375, |
| "completions/min_length": 9970.0, |
| "completions/min_terminated_length": 9970.0, |
| "entropy": 0.861279807984829, |
| "epoch": 0.015576503046327165, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008018149062991142, |
| "learning_rate": 1e-05, |
| "loss": 0.0669, |
| "num_tokens": 12567996.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999126195907593, |
| "sampling/importance_sampling_ratio/min": 0.10851297527551651, |
| "sampling/sampling_logp_difference/max": 2.2208855152130127, |
| "sampling/sampling_logp_difference/mean": 0.02991190180182457, |
| "step": 271 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002770495702861808, |
| "clip_ratio/high_mean": 0.0002770495702861808, |
| "clip_ratio/low_mean": 5.847953070770018e-05, |
| "clip_ratio/low_min": 5.847953070770018e-05, |
| "clip_ratio/region_mean": 0.000335529100993881, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10993.0, |
| "completions/max_terminated_length": 10993.0, |
| "completions/mean_length": 4628.25, |
| "completions/mean_terminated_length": 4628.25, |
| "completions/min_length": 2676.0, |
| "completions/min_terminated_length": 2676.0, |
| "entropy": 0.5769103765487671, |
| "epoch": 0.015633980917346822, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010209813714027405, |
| "learning_rate": 1e-05, |
| "loss": -0.0269, |
| "num_tokens": 12606310.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0006463527679443, |
| "sampling/importance_sampling_ratio/min": 0.41931721568107605, |
| "sampling/sampling_logp_difference/max": 0.869127631187439, |
| "sampling/sampling_logp_difference/mean": 0.0236323531717062, |
| "step": 272 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5765.0, |
| "completions/max_terminated_length": 5765.0, |
| "completions/mean_length": 5065.125, |
| "completions/mean_terminated_length": 5065.125, |
| "completions/min_length": 4460.0, |
| "completions/min_terminated_length": 4460.0, |
| "entropy": 0.36210478469729424, |
| "epoch": 0.015691458788366478, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12647839.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999702572822571, |
| "sampling/importance_sampling_ratio/min": 0.605582058429718, |
| "sampling/sampling_logp_difference/max": 0.7214326858520508, |
| "sampling/sampling_logp_difference/mean": 0.014195986092090607, |
| "step": 273 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 5.6531243899371475e-05, |
| "clip_ratio/low_min": 5.6531243899371475e-05, |
| "clip_ratio/region_mean": 5.6531243899371475e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13267.0, |
| "completions/max_terminated_length": 13267.0, |
| "completions/mean_length": 4615.625, |
| "completions/mean_terminated_length": 4615.625, |
| "completions/min_length": 2352.0, |
| "completions/min_terminated_length": 2352.0, |
| "entropy": 0.33000186271965504, |
| "epoch": 0.015748936659386138, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.12380556017160416, |
| "learning_rate": 1e-05, |
| "loss": 0.6623, |
| "num_tokens": 12685988.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000193476676941, |
| "sampling/importance_sampling_ratio/min": 0.541982889175415, |
| "sampling/sampling_logp_difference/max": 1.0451645851135254, |
| "sampling/sampling_logp_difference/mean": 0.01526904758065939, |
| "step": 274 |
| }, |
| { |
| "clip_ratio/high_max": 3.398894114070572e-05, |
| "clip_ratio/high_mean": 3.398894114070572e-05, |
| "clip_ratio/low_mean": 0.0008203986872103997, |
| "clip_ratio/low_min": 0.0008203986872103997, |
| "clip_ratio/region_mean": 0.0008543876283511054, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16029.0, |
| "completions/mean_length": 13501.0, |
| "completions/mean_terminated_length": 11771.2001953125, |
| "completions/min_length": 7672.0, |
| "completions/min_terminated_length": 7672.0, |
| "entropy": 0.6330981440842152, |
| "epoch": 0.015806414530405793, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.011000245809555054, |
| "learning_rate": 1e-05, |
| "loss": 0.1659, |
| "num_tokens": 12795020.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998465180397034, |
| "sampling/importance_sampling_ratio/min": 1.6777634073150693e-06, |
| "sampling/sampling_logp_difference/max": 13.298048973083496, |
| "sampling/sampling_logp_difference/mean": 0.02542637661099434, |
| "step": 275 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5439.0, |
| "completions/max_terminated_length": 5439.0, |
| "completions/mean_length": 2809.625, |
| "completions/mean_terminated_length": 2809.625, |
| "completions/min_length": 1897.0, |
| "completions/min_terminated_length": 1897.0, |
| "entropy": 0.2729002833366394, |
| "epoch": 0.015863892401425453, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12818761.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4460861682891846, |
| "sampling/importance_sampling_ratio/mean": 1.000095009803772, |
| "sampling/importance_sampling_ratio/min": 0.47927430272102356, |
| "sampling/sampling_logp_difference/max": 0.7354822158813477, |
| "sampling/sampling_logp_difference/mean": 0.012092736549675465, |
| "step": 276 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5087.0, |
| "completions/max_terminated_length": 5087.0, |
| "completions/mean_length": 3614.5, |
| "completions/mean_terminated_length": 3614.5, |
| "completions/min_length": 3021.0, |
| "completions/min_terminated_length": 3021.0, |
| "entropy": 0.47840742766857147, |
| "epoch": 0.01592137027244511, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 12848629.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.450257658958435, |
| "sampling/importance_sampling_ratio/mean": 0.9998756051063538, |
| "sampling/importance_sampling_ratio/min": 0.6375133395195007, |
| "sampling/sampling_logp_difference/max": 0.4501800537109375, |
| "sampling/sampling_logp_difference/mean": 0.013919536024332047, |
| "step": 277 |
| }, |
| { |
| "clip_ratio/high_max": 3.3866159355966374e-05, |
| "clip_ratio/high_mean": 3.3866159355966374e-05, |
| "clip_ratio/low_mean": 0.0006137364034657367, |
| "clip_ratio/low_min": 0.0006137364034657367, |
| "clip_ratio/region_mean": 0.000647602562821703, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 11073.0, |
| "completions/mean_length": 10755.125, |
| "completions/mean_terminated_length": 7377.80029296875, |
| "completions/min_length": 2662.0, |
| "completions/min_terminated_length": 2662.0, |
| "entropy": 0.5293156765401363, |
| "epoch": 0.015978848143464765, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01830984279513359, |
| "learning_rate": 1e-05, |
| "loss": 0.3361, |
| "num_tokens": 12936262.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0003126859664917, |
| "sampling/importance_sampling_ratio/min": 0.20826758444309235, |
| "sampling/sampling_logp_difference/max": 1.8035707473754883, |
| "sampling/sampling_logp_difference/mean": 0.023199284449219704, |
| "step": 278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 2.079520891129505e-05, |
| "clip_ratio/low_min": 2.079520891129505e-05, |
| "clip_ratio/region_mean": 2.079520891129505e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6011.0, |
| "completions/max_terminated_length": 6011.0, |
| "completions/mean_length": 4201.625, |
| "completions/mean_terminated_length": 4201.625, |
| "completions/min_length": 3057.0, |
| "completions/min_terminated_length": 3057.0, |
| "entropy": 0.1944421762600541, |
| "epoch": 0.016036326014484424, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005797348450869322, |
| "learning_rate": 1e-05, |
| "loss": 0.1522, |
| "num_tokens": 12970995.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.550799012184143, |
| "sampling/importance_sampling_ratio/mean": 1.0001014471054077, |
| "sampling/importance_sampling_ratio/min": 0.5867496132850647, |
| "sampling/sampling_logp_difference/max": 0.5331571102142334, |
| "sampling/sampling_logp_difference/mean": 0.00934829842299223, |
| "step": 279 |
| }, |
| { |
| "clip_ratio/high_max": 0.0001857445422501769, |
| "clip_ratio/high_mean": 0.0001857445422501769, |
| "clip_ratio/low_mean": 5.3636558732250705e-05, |
| "clip_ratio/low_min": 5.3636558732250705e-05, |
| "clip_ratio/region_mean": 0.0002393811009824276, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9322.0, |
| "completions/max_terminated_length": 9322.0, |
| "completions/mean_length": 5055.5, |
| "completions/mean_terminated_length": 5055.5, |
| "completions/min_length": 2883.0, |
| "completions/min_terminated_length": 2883.0, |
| "entropy": 0.5680248625576496, |
| "epoch": 0.01609380388550408, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0072536226361989975, |
| "learning_rate": 1e-05, |
| "loss": 0.2987, |
| "num_tokens": 13014351.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.4945365190505981, |
| "sampling/importance_sampling_ratio/mean": 0.9998281002044678, |
| "sampling/importance_sampling_ratio/min": 0.558586061000824, |
| "sampling/sampling_logp_difference/max": 0.5823465585708618, |
| "sampling/sampling_logp_difference/mean": 0.02077416516840458, |
| "step": 280 |
| }, |
| { |
| "clip_ratio/high_max": 4.7874134907033294e-05, |
| "clip_ratio/high_mean": 4.7874134907033294e-05, |
| "clip_ratio/low_mean": 0.0006891895754961297, |
| "clip_ratio/low_min": 0.0006891895754961297, |
| "clip_ratio/region_mean": 0.000737063710403163, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9291.0, |
| "completions/max_terminated_length": 9291.0, |
| "completions/mean_length": 6910.375, |
| "completions/mean_terminated_length": 6910.375, |
| "completions/min_length": 1861.0, |
| "completions/min_terminated_length": 1861.0, |
| "entropy": 0.587644200772047, |
| "epoch": 0.01615128175652374, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.019552532583475113, |
| "learning_rate": 1e-05, |
| "loss": -0.1338, |
| "num_tokens": 13070618.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999970197677612, |
| "sampling/importance_sampling_ratio/min": 0.4577460289001465, |
| "sampling/sampling_logp_difference/max": 0.7814407348632812, |
| "sampling/sampling_logp_difference/mean": 0.02277231030166149, |
| "step": 281 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 1.6823687474243343e-05, |
| "clip_ratio/low_min": 1.6823687474243343e-05, |
| "clip_ratio/region_mean": 1.6823687474243343e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7430.0, |
| "completions/max_terminated_length": 7430.0, |
| "completions/mean_length": 3620.375, |
| "completions/mean_terminated_length": 3620.375, |
| "completions/min_length": 1958.0, |
| "completions/min_terminated_length": 1958.0, |
| "entropy": 0.41035582683980465, |
| "epoch": 0.016208759627543395, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005621267948299646, |
| "learning_rate": 1e-05, |
| "loss": 0.3721, |
| "num_tokens": 13100749.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6037373542785645, |
| "sampling/importance_sampling_ratio/mean": 1.0002422332763672, |
| "sampling/importance_sampling_ratio/min": 0.6739776134490967, |
| "sampling/sampling_logp_difference/max": 0.4723367691040039, |
| "sampling/sampling_logp_difference/mean": 0.012744064442813396, |
| "step": 282 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12909.0, |
| "completions/max_terminated_length": 12909.0, |
| "completions/mean_length": 7139.75, |
| "completions/mean_terminated_length": 7139.75, |
| "completions/min_length": 3028.0, |
| "completions/min_terminated_length": 3028.0, |
| "entropy": 0.3640165962278843, |
| "epoch": 0.016266237498563055, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13159011.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8976774215698242, |
| "sampling/importance_sampling_ratio/mean": 0.9998587369918823, |
| "sampling/importance_sampling_ratio/min": 0.5414568185806274, |
| "sampling/sampling_logp_difference/max": 0.6406307220458984, |
| "sampling/sampling_logp_difference/mean": 0.015155038796365261, |
| "step": 283 |
| }, |
| { |
| "clip_ratio/high_max": 8.563771370972972e-05, |
| "clip_ratio/high_mean": 8.563771370972972e-05, |
| "clip_ratio/low_mean": 0.00022893887216923758, |
| "clip_ratio/low_min": 0.00022893887216923758, |
| "clip_ratio/region_mean": 0.0003145765858789673, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11161.0, |
| "completions/max_terminated_length": 11161.0, |
| "completions/mean_length": 7993.25, |
| "completions/mean_terminated_length": 7993.25, |
| "completions/min_length": 4849.0, |
| "completions/min_terminated_length": 4849.0, |
| "entropy": 0.6444938853383064, |
| "epoch": 0.01632371536958271, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.045867763459682465, |
| "learning_rate": 1e-05, |
| "loss": 0.1413, |
| "num_tokens": 13223757.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999954879283905, |
| "sampling/importance_sampling_ratio/min": 0.10870083421468735, |
| "sampling/sampling_logp_difference/max": 2.219155788421631, |
| "sampling/sampling_logp_difference/mean": 0.024017199873924255, |
| "step": 284 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4588.0, |
| "completions/max_terminated_length": 4588.0, |
| "completions/mean_length": 2995.125, |
| "completions/mean_terminated_length": 2995.125, |
| "completions/min_length": 1353.0, |
| "completions/min_terminated_length": 1353.0, |
| "entropy": 0.38860574923455715, |
| "epoch": 0.016381193240602367, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13248526.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9996554255485535, |
| "sampling/importance_sampling_ratio/min": 0.5708932876586914, |
| "sampling/sampling_logp_difference/max": 0.7145235538482666, |
| "sampling/sampling_logp_difference/mean": 0.016480296850204468, |
| "step": 285 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15923.0, |
| "completions/mean_length": 14015.375, |
| "completions/mean_terminated_length": 13225.833984375, |
| "completions/min_length": 11374.0, |
| "completions/min_terminated_length": 11374.0, |
| "entropy": 0.4670153819024563, |
| "epoch": 0.016438671111622026, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13361801.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002036094665527, |
| "sampling/importance_sampling_ratio/min": 0.39538297057151794, |
| "sampling/sampling_logp_difference/max": 0.9279004335403442, |
| "sampling/sampling_logp_difference/mean": 0.018330631777644157, |
| "step": 286 |
| }, |
| { |
| "clip_ratio/high_max": 6.584697439393494e-05, |
| "clip_ratio/high_mean": 6.584697439393494e-05, |
| "clip_ratio/low_mean": 4.254305531503633e-05, |
| "clip_ratio/low_min": 4.254305531503633e-05, |
| "clip_ratio/region_mean": 0.00010839002970897127, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14691.0, |
| "completions/max_terminated_length": 14691.0, |
| "completions/mean_length": 8030.25, |
| "completions/mean_terminated_length": 8030.25, |
| "completions/min_length": 2975.0, |
| "completions/min_terminated_length": 2975.0, |
| "entropy": 0.6689124628901482, |
| "epoch": 0.016496148982641682, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005533882882446051, |
| "learning_rate": 1e-05, |
| "loss": 0.2931, |
| "num_tokens": 13427291.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000213384628296, |
| "sampling/importance_sampling_ratio/min": 0.22132240235805511, |
| "sampling/sampling_logp_difference/max": 1.5081348419189453, |
| "sampling/sampling_logp_difference/mean": 0.020459897816181183, |
| "step": 287 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0001402131310896948, |
| "clip_ratio/low_min": 0.0001402131310896948, |
| "clip_ratio/region_mean": 0.0001402131310896948, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3510.0, |
| "completions/max_terminated_length": 3510.0, |
| "completions/mean_length": 1911.125, |
| "completions/mean_terminated_length": 1911.125, |
| "completions/min_length": 1122.0, |
| "completions/min_terminated_length": 1122.0, |
| "entropy": 0.40920743718743324, |
| "epoch": 0.01655362685366134, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01969943195581436, |
| "learning_rate": 1e-05, |
| "loss": -0.0788, |
| "num_tokens": 13443700.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.2755016088485718, |
| "sampling/importance_sampling_ratio/mean": 1.0000691413879395, |
| "sampling/importance_sampling_ratio/min": 0.6171529293060303, |
| "sampling/sampling_logp_difference/max": 0.4826383590698242, |
| "sampling/sampling_logp_difference/mean": 0.013039803132414818, |
| "step": 288 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3207.0, |
| "completions/max_terminated_length": 3207.0, |
| "completions/mean_length": 2260.0, |
| "completions/mean_terminated_length": 2260.0, |
| "completions/min_length": 1123.0, |
| "completions/min_terminated_length": 1123.0, |
| "entropy": 0.24728919006884098, |
| "epoch": 0.016611104724680997, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008269250392913818, |
| "learning_rate": 1e-05, |
| "loss": 0.0429, |
| "num_tokens": 13463084.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.536453127861023, |
| "sampling/importance_sampling_ratio/mean": 1.0002949237823486, |
| "sampling/importance_sampling_ratio/min": 0.6286458969116211, |
| "sampling/sampling_logp_difference/max": 0.4641871452331543, |
| "sampling/sampling_logp_difference/mean": 0.008462048135697842, |
| "step": 289 |
| }, |
| { |
| "clip_ratio/high_max": 8.435608469881117e-05, |
| "clip_ratio/high_mean": 8.435608469881117e-05, |
| "clip_ratio/low_mean": 0.0005368705315049738, |
| "clip_ratio/low_min": 0.0005368705315049738, |
| "clip_ratio/region_mean": 0.0006212266162037849, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12413.0, |
| "completions/mean_length": 10500.5, |
| "completions/mean_terminated_length": 8539.333984375, |
| "completions/min_length": 5737.0, |
| "completions/min_terminated_length": 5737.0, |
| "entropy": 0.39540280401706696, |
| "epoch": 0.016668582595700657, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014503099955618382, |
| "learning_rate": 1e-05, |
| "loss": 0.106, |
| "num_tokens": 13548144.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0002013444900513, |
| "sampling/importance_sampling_ratio/min": 0.4272690713405609, |
| "sampling/sampling_logp_difference/max": 0.9638075828552246, |
| "sampling/sampling_logp_difference/mean": 0.017945846542716026, |
| "step": 290 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3095.0, |
| "completions/max_terminated_length": 3095.0, |
| "completions/mean_length": 2623.5, |
| "completions/mean_terminated_length": 2623.5, |
| "completions/min_length": 2247.0, |
| "completions/min_terminated_length": 2247.0, |
| "entropy": 0.19449901394546032, |
| "epoch": 0.016726060466720313, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13570308.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.380873441696167, |
| "sampling/importance_sampling_ratio/mean": 0.9999493956565857, |
| "sampling/importance_sampling_ratio/min": 0.6107633113861084, |
| "sampling/sampling_logp_difference/max": 0.4930458068847656, |
| "sampling/sampling_logp_difference/mean": 0.008187439292669296, |
| "step": 291 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012352219346212223, |
| "clip_ratio/high_mean": 0.00012352219346212223, |
| "clip_ratio/low_mean": 0.00023838504421291873, |
| "clip_ratio/low_min": 0.00023838504421291873, |
| "clip_ratio/region_mean": 0.00036190723767504096, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11632.0, |
| "completions/max_terminated_length": 11632.0, |
| "completions/mean_length": 5898.125, |
| "completions/mean_terminated_length": 5898.125, |
| "completions/min_length": 2433.0, |
| "completions/min_terminated_length": 2433.0, |
| "entropy": 0.3482863698154688, |
| "epoch": 0.01678353833773997, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05515585094690323, |
| "learning_rate": 1e-05, |
| "loss": 0.5142, |
| "num_tokens": 13619077.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999215602874756, |
| "sampling/importance_sampling_ratio/min": 0.47605910897254944, |
| "sampling/sampling_logp_difference/max": 0.742213249206543, |
| "sampling/sampling_logp_difference/mean": 0.016893424093723297, |
| "step": 292 |
| }, |
| { |
| "clip_ratio/high_max": 9.655231951910537e-05, |
| "clip_ratio/high_mean": 9.655231951910537e-05, |
| "clip_ratio/low_mean": 0.00034231648896820843, |
| "clip_ratio/low_min": 0.00034231648896820843, |
| "clip_ratio/region_mean": 0.0004388688084873138, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9685.0, |
| "completions/max_terminated_length": 9685.0, |
| "completions/mean_length": 4638.375, |
| "completions/mean_terminated_length": 4638.375, |
| "completions/min_length": 1932.0, |
| "completions/min_terminated_length": 1932.0, |
| "entropy": 0.5883205719292164, |
| "epoch": 0.016841016208759628, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.028066974133253098, |
| "learning_rate": 1e-05, |
| "loss": 0.1608, |
| "num_tokens": 13657504.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.80884850025177, |
| "sampling/importance_sampling_ratio/mean": 0.9993683099746704, |
| "sampling/importance_sampling_ratio/min": 0.388579785823822, |
| "sampling/sampling_logp_difference/max": 0.9452568292617798, |
| "sampling/sampling_logp_difference/mean": 0.021590445190668106, |
| "step": 293 |
| }, |
| { |
| "clip_ratio/high_max": 1.4988009752414655e-05, |
| "clip_ratio/high_mean": 1.4988009752414655e-05, |
| "clip_ratio/low_mean": 0.0007542960302089341, |
| "clip_ratio/low_min": 0.0007542960302089341, |
| "clip_ratio/region_mean": 0.0007692840399613488, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14972.0, |
| "completions/mean_length": 12955.125, |
| "completions/mean_terminated_length": 11812.1669921875, |
| "completions/min_length": 8340.0, |
| "completions/min_terminated_length": 8340.0, |
| "entropy": 0.6050230599939823, |
| "epoch": 0.016898494079779284, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005949932616204023, |
| "learning_rate": 1e-05, |
| "loss": 0.1258, |
| "num_tokens": 13762497.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999903678894043, |
| "sampling/importance_sampling_ratio/min": 0.02129349298775196, |
| "sampling/sampling_logp_difference/max": 3.849353790283203, |
| "sampling/sampling_logp_difference/mean": 0.022619644179940224, |
| "step": 294 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10462.0, |
| "completions/max_terminated_length": 10462.0, |
| "completions/mean_length": 7412.0, |
| "completions/mean_terminated_length": 7412.0, |
| "completions/min_length": 5436.0, |
| "completions/min_terminated_length": 5436.0, |
| "entropy": 0.6282271333038807, |
| "epoch": 0.016955971950798943, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13822553.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999313354492188, |
| "sampling/importance_sampling_ratio/min": 0.3860255777835846, |
| "sampling/sampling_logp_difference/max": 0.9518516063690186, |
| "sampling/sampling_logp_difference/mean": 0.02337348833680153, |
| "step": 295 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7661.0, |
| "completions/max_terminated_length": 7661.0, |
| "completions/mean_length": 5066.875, |
| "completions/mean_terminated_length": 5066.875, |
| "completions/min_length": 2909.0, |
| "completions/min_terminated_length": 2909.0, |
| "entropy": 0.41942763701081276, |
| "epoch": 0.0170134498218186, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 13864168.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6161707639694214, |
| "sampling/importance_sampling_ratio/mean": 1.0000927448272705, |
| "sampling/importance_sampling_ratio/min": 0.6308397650718689, |
| "sampling/sampling_logp_difference/max": 0.4800596237182617, |
| "sampling/sampling_logp_difference/mean": 0.016679463908076286, |
| "step": 296 |
| }, |
| { |
| "clip_ratio/high_max": 2.8115160603192635e-05, |
| "clip_ratio/high_mean": 2.8115160603192635e-05, |
| "clip_ratio/low_mean": 0.00034843457979150116, |
| "clip_ratio/low_min": 0.00034843457979150116, |
| "clip_ratio/region_mean": 0.0003765497403946938, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5227.0, |
| "completions/max_terminated_length": 5227.0, |
| "completions/mean_length": 3872.875, |
| "completions/mean_terminated_length": 3872.875, |
| "completions/min_length": 2402.0, |
| "completions/min_terminated_length": 2402.0, |
| "entropy": 0.456345247104764, |
| "epoch": 0.01707092769283826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.04224399849772453, |
| "learning_rate": 1e-05, |
| "loss": 0.1888, |
| "num_tokens": 13896335.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.3437304496765137, |
| "sampling/importance_sampling_ratio/mean": 0.9995068311691284, |
| "sampling/importance_sampling_ratio/min": 0.5417106747627258, |
| "sampling/sampling_logp_difference/max": 0.6130232810974121, |
| "sampling/sampling_logp_difference/mean": 0.016502318903803825, |
| "step": 297 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15812.0, |
| "completions/mean_length": 14368.375, |
| "completions/mean_terminated_length": 12352.75, |
| "completions/min_length": 6324.0, |
| "completions/min_terminated_length": 6324.0, |
| "entropy": 0.5045627877116203, |
| "epoch": 0.017128405563857915, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14012450.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000554323196411, |
| "sampling/importance_sampling_ratio/min": 0.2854660153388977, |
| "sampling/sampling_logp_difference/max": 1.2536323070526123, |
| "sampling/sampling_logp_difference/mean": 0.023292748257517815, |
| "step": 298 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1333.0, |
| "completions/max_terminated_length": 1333.0, |
| "completions/mean_length": 992.875, |
| "completions/mean_terminated_length": 992.875, |
| "completions/min_length": 658.0, |
| "completions/min_terminated_length": 658.0, |
| "entropy": 0.31343047320842743, |
| "epoch": 0.01718588343487757, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14021705.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7160178422927856, |
| "sampling/importance_sampling_ratio/mean": 0.999776303768158, |
| "sampling/importance_sampling_ratio/min": 0.6452099680900574, |
| "sampling/sampling_logp_difference/max": 0.5400063991546631, |
| "sampling/sampling_logp_difference/mean": 0.014022443443536758, |
| "step": 299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3130.0, |
| "completions/max_terminated_length": 3130.0, |
| "completions/mean_length": 2598.25, |
| "completions/mean_terminated_length": 2598.25, |
| "completions/min_length": 2000.0, |
| "completions/min_terminated_length": 2000.0, |
| "entropy": 0.3396691009402275, |
| "epoch": 0.01724336130589723, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14043851.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8454477787017822, |
| "sampling/importance_sampling_ratio/mean": 1.0004243850708008, |
| "sampling/importance_sampling_ratio/min": 0.676132321357727, |
| "sampling/sampling_logp_difference/max": 0.6127219200134277, |
| "sampling/sampling_logp_difference/mean": 0.01019161008298397, |
| "step": 300 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6920.0, |
| "completions/max_terminated_length": 6920.0, |
| "completions/mean_length": 4282.625, |
| "completions/mean_terminated_length": 4282.625, |
| "completions/min_length": 1837.0, |
| "completions/min_terminated_length": 1837.0, |
| "entropy": 1.0960607826709747, |
| "epoch": 0.017300839176916886, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14079664.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.480582356452942, |
| "sampling/importance_sampling_ratio/mean": 1.000103235244751, |
| "sampling/importance_sampling_ratio/min": 0.6124796271324158, |
| "sampling/sampling_logp_difference/max": 0.49023962020874023, |
| "sampling/sampling_logp_difference/mean": 0.026110351085662842, |
| "step": 301 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2898.0, |
| "completions/max_terminated_length": 2898.0, |
| "completions/mean_length": 2414.375, |
| "completions/mean_terminated_length": 2414.375, |
| "completions/min_length": 1927.0, |
| "completions/min_terminated_length": 1927.0, |
| "entropy": 0.23792114667594433, |
| "epoch": 0.017358317047936545, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14099739.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2692124843597412, |
| "sampling/importance_sampling_ratio/mean": 0.9998177886009216, |
| "sampling/importance_sampling_ratio/min": 0.6417900323867798, |
| "sampling/sampling_logp_difference/max": 0.4434940814971924, |
| "sampling/sampling_logp_difference/mean": 0.009595055133104324, |
| "step": 302 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15818.0, |
| "completions/max_terminated_length": 15818.0, |
| "completions/mean_length": 8399.25, |
| "completions/mean_terminated_length": 8399.25, |
| "completions/min_length": 3962.0, |
| "completions/min_terminated_length": 3962.0, |
| "entropy": 0.38004388101398945, |
| "epoch": 0.0174157949189562, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14168077.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998432397842407, |
| "sampling/importance_sampling_ratio/min": 0.4143325090408325, |
| "sampling/sampling_logp_difference/max": 0.8810864686965942, |
| "sampling/sampling_logp_difference/mean": 0.01728796400129795, |
| "step": 303 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13862.0, |
| "completions/mean_length": 14596.875, |
| "completions/mean_terminated_length": 12809.75, |
| "completions/min_length": 12360.0, |
| "completions/min_terminated_length": 12360.0, |
| "entropy": 0.520825169980526, |
| "epoch": 0.01747327278997586, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14286556.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998247027397156, |
| "sampling/importance_sampling_ratio/min": 0.0023807960096746683, |
| "sampling/sampling_logp_difference/max": 6.04032039642334, |
| "sampling/sampling_logp_difference/mean": 0.02044062502682209, |
| "step": 304 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3011.0, |
| "completions/max_terminated_length": 3011.0, |
| "completions/mean_length": 1954.125, |
| "completions/mean_terminated_length": 1954.125, |
| "completions/min_length": 1091.0, |
| "completions/min_terminated_length": 1091.0, |
| "entropy": 1.3233292549848557, |
| "epoch": 0.017530750660995516, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.05288930609822273, |
| "learning_rate": 1e-05, |
| "loss": -0.1429, |
| "num_tokens": 14303885.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.3491226434707642, |
| "sampling/importance_sampling_ratio/mean": 1.0000087022781372, |
| "sampling/importance_sampling_ratio/min": 0.6237119436264038, |
| "sampling/sampling_logp_difference/max": 0.47206664085388184, |
| "sampling/sampling_logp_difference/mean": 0.02923406846821308, |
| "step": 305 |
| }, |
| { |
| "clip_ratio/high_max": 9.506751302978955e-05, |
| "clip_ratio/high_mean": 9.506751302978955e-05, |
| "clip_ratio/low_mean": 0.00028230962925590575, |
| "clip_ratio/low_min": 0.00028230962925590575, |
| "clip_ratio/region_mean": 0.0003773771422856953, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 14650.0, |
| "completions/max_terminated_length": 14650.0, |
| "completions/mean_length": 6245.0, |
| "completions/mean_terminated_length": 6245.0, |
| "completions/min_length": 971.0, |
| "completions/min_terminated_length": 971.0, |
| "entropy": 0.5631169006228447, |
| "epoch": 0.017588228532015176, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03470282256603241, |
| "learning_rate": 1e-05, |
| "loss": -0.0158, |
| "num_tokens": 14355293.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000054121017456, |
| "sampling/importance_sampling_ratio/min": 0.4764721989631653, |
| "sampling/sampling_logp_difference/max": 0.8508948087692261, |
| "sampling/sampling_logp_difference/mean": 0.02263343334197998, |
| "step": 306 |
| }, |
| { |
| "clip_ratio/high_max": 6.634154306084383e-05, |
| "clip_ratio/high_mean": 6.634154306084383e-05, |
| "clip_ratio/low_mean": 8.58442799653858e-05, |
| "clip_ratio/low_min": 8.58442799653858e-05, |
| "clip_ratio/region_mean": 0.00015218582302622963, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11649.0, |
| "completions/max_terminated_length": 11649.0, |
| "completions/mean_length": 6025.0, |
| "completions/mean_terminated_length": 6025.0, |
| "completions/min_length": 1995.0, |
| "completions/min_terminated_length": 1995.0, |
| "entropy": 0.3133881501853466, |
| "epoch": 0.017645706403034832, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.005668353755027056, |
| "learning_rate": 1e-05, |
| "loss": 0.3299, |
| "num_tokens": 14404573.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.944168210029602, |
| "sampling/importance_sampling_ratio/mean": 0.9998335242271423, |
| "sampling/importance_sampling_ratio/min": 0.5526868104934692, |
| "sampling/sampling_logp_difference/max": 0.6648342609405518, |
| "sampling/sampling_logp_difference/mean": 0.01425552275031805, |
| "step": 307 |
| }, |
| { |
| "clip_ratio/high_max": 4.29996543971356e-05, |
| "clip_ratio/high_mean": 4.29996543971356e-05, |
| "clip_ratio/low_mean": 9.245562250725925e-05, |
| "clip_ratio/low_min": 9.245562250725925e-05, |
| "clip_ratio/region_mean": 0.00013545527690439485, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2980.0, |
| "completions/max_terminated_length": 2980.0, |
| "completions/mean_length": 1745.75, |
| "completions/mean_terminated_length": 1745.75, |
| "completions/min_length": 930.0, |
| "completions/min_terminated_length": 930.0, |
| "entropy": 0.30218998342752457, |
| "epoch": 0.017703184274054488, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01708727888762951, |
| "learning_rate": 1e-05, |
| "loss": -0.0797, |
| "num_tokens": 14419387.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6625542640686035, |
| "sampling/importance_sampling_ratio/mean": 1.0002760887145996, |
| "sampling/importance_sampling_ratio/min": 0.628564715385437, |
| "sampling/sampling_logp_difference/max": 0.5083551406860352, |
| "sampling/sampling_logp_difference/mean": 0.013656601309776306, |
| "step": 308 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11742.0, |
| "completions/max_terminated_length": 11742.0, |
| "completions/mean_length": 8171.25, |
| "completions/mean_terminated_length": 8171.25, |
| "completions/min_length": 4699.0, |
| "completions/min_terminated_length": 4699.0, |
| "entropy": 0.5202376432716846, |
| "epoch": 0.017760662145074147, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14485589.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8411750793457031, |
| "sampling/importance_sampling_ratio/mean": 1.0003048181533813, |
| "sampling/importance_sampling_ratio/min": 0.4910643994808197, |
| "sampling/sampling_logp_difference/max": 0.7111799716949463, |
| "sampling/sampling_logp_difference/mean": 0.020186688750982285, |
| "step": 309 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3891.0, |
| "completions/max_terminated_length": 3891.0, |
| "completions/mean_length": 1940.25, |
| "completions/mean_terminated_length": 1940.25, |
| "completions/min_length": 642.0, |
| "completions/min_terminated_length": 642.0, |
| "entropy": 0.31665168702602386, |
| "epoch": 0.017818140016093803, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14501911.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4268609285354614, |
| "sampling/importance_sampling_ratio/mean": 0.9999961256980896, |
| "sampling/importance_sampling_ratio/min": 0.6249384880065918, |
| "sampling/sampling_logp_difference/max": 0.47010207176208496, |
| "sampling/sampling_logp_difference/mean": 0.013832678087055683, |
| "step": 310 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6001.0, |
| "completions/max_terminated_length": 6001.0, |
| "completions/mean_length": 2945.5, |
| "completions/mean_terminated_length": 2945.5, |
| "completions/min_length": 1819.0, |
| "completions/min_terminated_length": 1819.0, |
| "entropy": 0.20620127767324448, |
| "epoch": 0.017875617887113462, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14526491.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3218950033187866, |
| "sampling/importance_sampling_ratio/mean": 0.9999508857727051, |
| "sampling/importance_sampling_ratio/min": 0.6079448461532593, |
| "sampling/sampling_logp_difference/max": 0.49767112731933594, |
| "sampling/sampling_logp_difference/mean": 0.010383360087871552, |
| "step": 311 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3221.0, |
| "completions/max_terminated_length": 3221.0, |
| "completions/mean_length": 1896.25, |
| "completions/mean_terminated_length": 1896.25, |
| "completions/min_length": 1382.0, |
| "completions/min_terminated_length": 1382.0, |
| "entropy": 0.3582219146192074, |
| "epoch": 0.01793309575813312, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14542309.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5014512538909912, |
| "sampling/importance_sampling_ratio/mean": 1.0002200603485107, |
| "sampling/importance_sampling_ratio/min": 0.647032618522644, |
| "sampling/sampling_logp_difference/max": 0.43535852432250977, |
| "sampling/sampling_logp_difference/mean": 0.015400128439068794, |
| "step": 312 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6912.0, |
| "completions/max_terminated_length": 6912.0, |
| "completions/mean_length": 4067.125, |
| "completions/mean_terminated_length": 4067.125, |
| "completions/min_length": 685.0, |
| "completions/min_terminated_length": 685.0, |
| "entropy": 0.517782075330615, |
| "epoch": 0.017990573629152778, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14575734.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.418588399887085, |
| "sampling/importance_sampling_ratio/mean": 1.0005604028701782, |
| "sampling/importance_sampling_ratio/min": 0.5757678151130676, |
| "sampling/sampling_logp_difference/max": 0.5520508289337158, |
| "sampling/sampling_logp_difference/mean": 0.022016707807779312, |
| "step": 313 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5348.0, |
| "completions/max_terminated_length": 5348.0, |
| "completions/mean_length": 3690.625, |
| "completions/mean_terminated_length": 3690.625, |
| "completions/min_length": 2375.0, |
| "completions/min_terminated_length": 2375.0, |
| "entropy": 0.289380444213748, |
| "epoch": 0.018048051500172434, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14606091.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4627519845962524, |
| "sampling/importance_sampling_ratio/mean": 0.9998953938484192, |
| "sampling/importance_sampling_ratio/min": 0.6064385771751404, |
| "sampling/sampling_logp_difference/max": 0.5001518726348877, |
| "sampling/sampling_logp_difference/mean": 0.01178512629121542, |
| "step": 314 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10975.0, |
| "completions/max_terminated_length": 10975.0, |
| "completions/mean_length": 4978.25, |
| "completions/mean_terminated_length": 4978.25, |
| "completions/min_length": 2371.0, |
| "completions/min_terminated_length": 2371.0, |
| "entropy": 0.5141785815358162, |
| "epoch": 0.01810552937119209, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14647109.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5613679885864258, |
| "sampling/importance_sampling_ratio/mean": 0.9998388290405273, |
| "sampling/importance_sampling_ratio/min": 0.4340689778327942, |
| "sampling/sampling_logp_difference/max": 0.8345518112182617, |
| "sampling/sampling_logp_difference/mean": 0.021983593702316284, |
| "step": 315 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00018217236720374785, |
| "clip_ratio/low_min": 0.00018217236720374785, |
| "clip_ratio/region_mean": 0.00018217236720374785, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7029.0, |
| "completions/max_terminated_length": 7029.0, |
| "completions/mean_length": 5591.375, |
| "completions/mean_terminated_length": 5591.375, |
| "completions/min_length": 2976.0, |
| "completions/min_terminated_length": 2976.0, |
| "entropy": 0.7261018678545952, |
| "epoch": 0.01816300724221175, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009648680686950684, |
| "learning_rate": 1e-05, |
| "loss": 0.1141, |
| "num_tokens": 14693080.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.7077374458312988, |
| "sampling/importance_sampling_ratio/mean": 0.9998070001602173, |
| "sampling/importance_sampling_ratio/min": 0.5865379571914673, |
| "sampling/sampling_logp_difference/max": 0.5351693630218506, |
| "sampling/sampling_logp_difference/mean": 0.020792080089449883, |
| "step": 316 |
| }, |
| { |
| "clip_ratio/high_max": 1.1924067621293943e-05, |
| "clip_ratio/high_mean": 1.1924067621293943e-05, |
| "clip_ratio/low_mean": 0.00020796335957129486, |
| "clip_ratio/low_min": 0.00020796335957129486, |
| "clip_ratio/region_mean": 0.0002198874271925888, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10483.0, |
| "completions/max_terminated_length": 10483.0, |
| "completions/mean_length": 3633.75, |
| "completions/mean_terminated_length": 3633.75, |
| "completions/min_length": 1952.0, |
| "completions/min_terminated_length": 1952.0, |
| "entropy": 0.3380902595818043, |
| "epoch": 0.018220485113231405, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007336276583373547, |
| "learning_rate": 1e-05, |
| "loss": -0.6667, |
| "num_tokens": 14723190.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0004078149795532, |
| "sampling/importance_sampling_ratio/min": 0.5002537369728088, |
| "sampling/sampling_logp_difference/max": 0.7221651077270508, |
| "sampling/sampling_logp_difference/mean": 0.013298140838742256, |
| "step": 317 |
| }, |
| { |
| "clip_ratio/high_max": 3.0288345442386344e-05, |
| "clip_ratio/high_mean": 3.0288345442386344e-05, |
| "clip_ratio/low_mean": 0.0007701888825977221, |
| "clip_ratio/low_min": 0.0007701888825977221, |
| "clip_ratio/region_mean": 0.0008004772280401085, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15778.0, |
| "completions/mean_length": 15147.375, |
| "completions/mean_terminated_length": 13910.75, |
| "completions/min_length": 12328.0, |
| "completions/min_terminated_length": 12328.0, |
| "entropy": 0.6852492205798626, |
| "epoch": 0.018277962984251064, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.006635353900492191, |
| "learning_rate": 1e-05, |
| "loss": 0.0648, |
| "num_tokens": 14845417.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998446106910706, |
| "sampling/importance_sampling_ratio/min": 0.031520646065473557, |
| "sampling/sampling_logp_difference/max": 3.4571125507354736, |
| "sampling/sampling_logp_difference/mean": 0.02578580379486084, |
| "step": 318 |
| }, |
| { |
| "clip_ratio/high_max": 0.00013082451914669946, |
| "clip_ratio/high_mean": 0.00013082451914669946, |
| "clip_ratio/low_mean": 0.00016318538109771907, |
| "clip_ratio/low_min": 0.00016318538109771907, |
| "clip_ratio/region_mean": 0.00029400990024441853, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3081.0, |
| "completions/max_terminated_length": 3081.0, |
| "completions/mean_length": 1920.375, |
| "completions/mean_terminated_length": 1920.375, |
| "completions/min_length": 766.0, |
| "completions/min_terminated_length": 766.0, |
| "entropy": 0.4040065184235573, |
| "epoch": 0.01833544085527072, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07099581509828568, |
| "learning_rate": 1e-05, |
| "loss": -0.2124, |
| "num_tokens": 14861540.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.299908995628357, |
| "sampling/importance_sampling_ratio/mean": 1.000104546546936, |
| "sampling/importance_sampling_ratio/min": 0.6175916194915771, |
| "sampling/sampling_logp_difference/max": 0.48192787170410156, |
| "sampling/sampling_logp_difference/mean": 0.01673056185245514, |
| "step": 319 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0002039020328084007, |
| "clip_ratio/low_min": 0.0002039020328084007, |
| "clip_ratio/region_mean": 0.0002039020328084007, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10435.0, |
| "completions/max_terminated_length": 10435.0, |
| "completions/mean_length": 7574.375, |
| "completions/mean_terminated_length": 7574.375, |
| "completions/min_length": 4115.0, |
| "completions/min_terminated_length": 4115.0, |
| "entropy": 0.6065731458365917, |
| "epoch": 0.01839291872629038, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.014942534267902374, |
| "learning_rate": 1e-05, |
| "loss": 0.0561, |
| "num_tokens": 14923455.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 1.6245927810668945, |
| "sampling/importance_sampling_ratio/mean": 0.9999759793281555, |
| "sampling/importance_sampling_ratio/min": 0.19265475869178772, |
| "sampling/sampling_logp_difference/max": 1.6468554735183716, |
| "sampling/sampling_logp_difference/mean": 0.018679451197385788, |
| "step": 320 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7008.0, |
| "completions/max_terminated_length": 7008.0, |
| "completions/mean_length": 4057.75, |
| "completions/mean_terminated_length": 4057.75, |
| "completions/min_length": 1055.0, |
| "completions/min_terminated_length": 1055.0, |
| "entropy": 0.3605603091418743, |
| "epoch": 0.018450396597310036, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 14956845.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001881122589111, |
| "sampling/importance_sampling_ratio/min": 0.5888962149620056, |
| "sampling/sampling_logp_difference/max": 0.8162369728088379, |
| "sampling/sampling_logp_difference/mean": 0.017093200236558914, |
| "step": 321 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14493.0, |
| "completions/mean_length": 12585.875, |
| "completions/mean_terminated_length": 11319.833984375, |
| "completions/min_length": 5750.0, |
| "completions/min_terminated_length": 5750.0, |
| "entropy": 0.8011277616024017, |
| "epoch": 0.01850787446832969, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15058364.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999358057975769, |
| "sampling/importance_sampling_ratio/min": 0.00016261611017398536, |
| "sampling/sampling_logp_difference/max": 8.72411823272705, |
| "sampling/sampling_logp_difference/mean": 0.026181455701589584, |
| "step": 322 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8791.0, |
| "completions/max_terminated_length": 8791.0, |
| "completions/mean_length": 4230.875, |
| "completions/mean_terminated_length": 4230.875, |
| "completions/min_length": 1668.0, |
| "completions/min_terminated_length": 1668.0, |
| "entropy": 0.38680815510451794, |
| "epoch": 0.01856535233934935, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15093627.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7237930297851562, |
| "sampling/importance_sampling_ratio/mean": 1.0000313520431519, |
| "sampling/importance_sampling_ratio/min": 0.5658870339393616, |
| "sampling/sampling_logp_difference/max": 0.5693607330322266, |
| "sampling/sampling_logp_difference/mean": 0.01748107559978962, |
| "step": 323 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5690.0, |
| "completions/max_terminated_length": 5690.0, |
| "completions/mean_length": 2776.875, |
| "completions/mean_terminated_length": 2776.875, |
| "completions/min_length": 1956.0, |
| "completions/min_terminated_length": 1956.0, |
| "entropy": 0.7585382387042046, |
| "epoch": 0.018622830210369007, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15117330.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.8585883378982544, |
| "sampling/importance_sampling_ratio/mean": 1.0003856420516968, |
| "sampling/importance_sampling_ratio/min": 0.6560837626457214, |
| "sampling/sampling_logp_difference/max": 0.6198172569274902, |
| "sampling/sampling_logp_difference/mean": 0.021034816280007362, |
| "step": 324 |
| }, |
| { |
| "clip_ratio/high_max": 8.878271182766184e-05, |
| "clip_ratio/high_mean": 8.878271182766184e-05, |
| "clip_ratio/low_mean": 9.278733341488987e-05, |
| "clip_ratio/low_min": 9.278733341488987e-05, |
| "clip_ratio/region_mean": 0.00018157004524255171, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8083.0, |
| "completions/max_terminated_length": 8083.0, |
| "completions/mean_length": 4144.125, |
| "completions/mean_terminated_length": 4144.125, |
| "completions/min_length": 488.0, |
| "completions/min_terminated_length": 488.0, |
| "entropy": 0.5541177727282047, |
| "epoch": 0.018680308081388666, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07104233652353287, |
| "learning_rate": 1e-05, |
| "loss": 0.0185, |
| "num_tokens": 15152315.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.7055503129959106, |
| "sampling/importance_sampling_ratio/mean": 1.00011146068573, |
| "sampling/importance_sampling_ratio/min": 0.3620549738407135, |
| "sampling/sampling_logp_difference/max": 1.0159592628479004, |
| "sampling/sampling_logp_difference/mean": 0.02261045016348362, |
| "step": 325 |
| }, |
| { |
| "clip_ratio/high_max": 4.524165706243366e-05, |
| "clip_ratio/high_mean": 4.524165706243366e-05, |
| "clip_ratio/low_mean": 0.000441106480138842, |
| "clip_ratio/low_min": 0.000441106480138842, |
| "clip_ratio/region_mean": 0.0004863481372012757, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15512.0, |
| "completions/mean_length": 12479.875, |
| "completions/mean_terminated_length": 11922.1435546875, |
| "completions/min_length": 9749.0, |
| "completions/min_terminated_length": 9749.0, |
| "entropy": 0.7386438101530075, |
| "epoch": 0.018737785952408322, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.016033122316002846, |
| "learning_rate": 1e-05, |
| "loss": -0.0465, |
| "num_tokens": 15253610.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000121593475342, |
| "sampling/importance_sampling_ratio/min": 0.37671196460723877, |
| "sampling/sampling_logp_difference/max": 0.9762744903564453, |
| "sampling/sampling_logp_difference/mean": 0.02541348896920681, |
| "step": 326 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4390.0, |
| "completions/max_terminated_length": 4390.0, |
| "completions/mean_length": 1495.875, |
| "completions/mean_terminated_length": 1495.875, |
| "completions/min_length": 564.0, |
| "completions/min_terminated_length": 564.0, |
| "entropy": 0.3768142517656088, |
| "epoch": 0.01879526382342798, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15267001.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2828130722045898, |
| "sampling/importance_sampling_ratio/mean": 1.0002108812332153, |
| "sampling/importance_sampling_ratio/min": 0.6261738538742065, |
| "sampling/sampling_logp_difference/max": 0.4681272506713867, |
| "sampling/sampling_logp_difference/mean": 0.018461311236023903, |
| "step": 327 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 1976.0, |
| "completions/max_terminated_length": 1976.0, |
| "completions/mean_length": 1566.5, |
| "completions/mean_terminated_length": 1566.5, |
| "completions/min_length": 837.0, |
| "completions/min_terminated_length": 837.0, |
| "entropy": 0.5904304720461369, |
| "epoch": 0.018852741694447638, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15280733.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3126988410949707, |
| "sampling/importance_sampling_ratio/mean": 0.9998319745063782, |
| "sampling/importance_sampling_ratio/min": 0.6127909421920776, |
| "sampling/sampling_logp_difference/max": 0.48973143100738525, |
| "sampling/sampling_logp_difference/mean": 0.01677585020661354, |
| "step": 328 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2742.0, |
| "completions/max_terminated_length": 2742.0, |
| "completions/mean_length": 2204.0, |
| "completions/mean_terminated_length": 2204.0, |
| "completions/min_length": 1336.0, |
| "completions/min_terminated_length": 1336.0, |
| "entropy": 0.3187308609485626, |
| "epoch": 0.018910219565467293, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15299893.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4487329721450806, |
| "sampling/importance_sampling_ratio/mean": 1.000268578529358, |
| "sampling/importance_sampling_ratio/min": 0.6670248508453369, |
| "sampling/sampling_logp_difference/max": 0.40492796897888184, |
| "sampling/sampling_logp_difference/mean": 0.01043415255844593, |
| "step": 329 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7615.0, |
| "completions/max_terminated_length": 7615.0, |
| "completions/mean_length": 4871.875, |
| "completions/mean_terminated_length": 4871.875, |
| "completions/min_length": 3167.0, |
| "completions/min_terminated_length": 3167.0, |
| "entropy": 0.39124204590916634, |
| "epoch": 0.018967697436486953, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15340076.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4911690950393677, |
| "sampling/importance_sampling_ratio/mean": 0.9998763203620911, |
| "sampling/importance_sampling_ratio/min": 0.4422745108604431, |
| "sampling/sampling_logp_difference/max": 0.8158245086669922, |
| "sampling/sampling_logp_difference/mean": 0.01586175337433815, |
| "step": 330 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.75, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 16316.0, |
| "completions/mean_length": 16359.25, |
| "completions/mean_terminated_length": 16285.0, |
| "completions/min_length": 16254.0, |
| "completions/min_terminated_length": 16254.0, |
| "entropy": 0.6022672429680824, |
| "epoch": 0.01902517530750661, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15472374.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999624490737915, |
| "sampling/importance_sampling_ratio/min": 0.27834710478782654, |
| "sampling/sampling_logp_difference/max": 1.278886318206787, |
| "sampling/sampling_logp_difference/mean": 0.02268480323255062, |
| "step": 331 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3686.0, |
| "completions/max_terminated_length": 3686.0, |
| "completions/mean_length": 3110.5, |
| "completions/mean_terminated_length": 3110.5, |
| "completions/min_length": 2322.0, |
| "completions/min_terminated_length": 2322.0, |
| "entropy": 0.4437192752957344, |
| "epoch": 0.019082653178526268, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15498058.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3044779300689697, |
| "sampling/importance_sampling_ratio/mean": 1.000067114830017, |
| "sampling/importance_sampling_ratio/min": 0.6156363487243652, |
| "sampling/sampling_logp_difference/max": 0.48509883880615234, |
| "sampling/sampling_logp_difference/mean": 0.015801699832081795, |
| "step": 332 |
| }, |
| { |
| "clip_ratio/high_max": 1.7615557226235978e-05, |
| "clip_ratio/high_mean": 1.7615557226235978e-05, |
| "clip_ratio/low_mean": 0.0006183970363053959, |
| "clip_ratio/low_min": 0.0006183970363053959, |
| "clip_ratio/region_mean": 0.0006360125935316319, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10678.0, |
| "completions/max_terminated_length": 10678.0, |
| "completions/mean_length": 7056.25, |
| "completions/mean_terminated_length": 7056.25, |
| "completions/min_length": 3294.0, |
| "completions/min_terminated_length": 3294.0, |
| "entropy": 0.7451484203338623, |
| "epoch": 0.019140131049545924, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010370392352342606, |
| "learning_rate": 1e-05, |
| "loss": -0.0022, |
| "num_tokens": 15555412.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997419118881226, |
| "sampling/importance_sampling_ratio/min": 0.4893128573894501, |
| "sampling/sampling_logp_difference/max": 0.7159855365753174, |
| "sampling/sampling_logp_difference/mean": 0.026656731963157654, |
| "step": 333 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 6.742178811691701e-05, |
| "clip_ratio/low_min": 6.742178811691701e-05, |
| "clip_ratio/region_mean": 6.742178811691701e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4934.0, |
| "completions/max_terminated_length": 4934.0, |
| "completions/mean_length": 3034.125, |
| "completions/mean_terminated_length": 3034.125, |
| "completions/min_length": 1231.0, |
| "completions/min_terminated_length": 1231.0, |
| "entropy": 0.5302278809249401, |
| "epoch": 0.019197608920565584, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.06457998603582382, |
| "learning_rate": 1e-05, |
| "loss": -0.1002, |
| "num_tokens": 15580781.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.501657485961914, |
| "sampling/importance_sampling_ratio/mean": 1.0004233121871948, |
| "sampling/importance_sampling_ratio/min": 0.6905239820480347, |
| "sampling/sampling_logp_difference/max": 0.4065694808959961, |
| "sampling/sampling_logp_difference/mean": 0.01687503792345524, |
| "step": 334 |
| }, |
| { |
| "clip_ratio/high_max": 3.468056456767954e-05, |
| "clip_ratio/high_mean": 3.468056456767954e-05, |
| "clip_ratio/low_mean": 0.0002981360230478458, |
| "clip_ratio/low_min": 0.0002981360230478458, |
| "clip_ratio/region_mean": 0.00033281658761552535, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15099.0, |
| "completions/mean_length": 10131.625, |
| "completions/mean_terminated_length": 9238.4287109375, |
| "completions/min_length": 4174.0, |
| "completions/min_terminated_length": 4174.0, |
| "entropy": 0.4654873348772526, |
| "epoch": 0.01925508679158524, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010863181203603745, |
| "learning_rate": 1e-05, |
| "loss": 0.1874, |
| "num_tokens": 15662874.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999262094497681, |
| "sampling/importance_sampling_ratio/min": 0.42790281772613525, |
| "sampling/sampling_logp_difference/max": 0.8488591909408569, |
| "sampling/sampling_logp_difference/mean": 0.021146049723029137, |
| "step": 335 |
| }, |
| { |
| "clip_ratio/high_max": 4.132231333642267e-05, |
| "clip_ratio/high_mean": 4.132231333642267e-05, |
| "clip_ratio/low_mean": 3.9316419133683667e-05, |
| "clip_ratio/low_min": 3.9316419133683667e-05, |
| "clip_ratio/region_mean": 8.063873247010633e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9538.0, |
| "completions/max_terminated_length": 9538.0, |
| "completions/mean_length": 5125.25, |
| "completions/mean_terminated_length": 5125.25, |
| "completions/min_length": 1507.0, |
| "completions/min_terminated_length": 1507.0, |
| "entropy": 0.7648974284529686, |
| "epoch": 0.0193125646626049, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.0052408394403755665, |
| "learning_rate": 1e-05, |
| "loss": 0.3051, |
| "num_tokens": 15704684.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6716387271881104, |
| "sampling/importance_sampling_ratio/mean": 1.000206470489502, |
| "sampling/importance_sampling_ratio/min": 0.5740965008735657, |
| "sampling/sampling_logp_difference/max": 0.5549578666687012, |
| "sampling/sampling_logp_difference/mean": 0.021035529673099518, |
| "step": 336 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0007146268180804327, |
| "clip_ratio/low_min": 0.0007146268180804327, |
| "clip_ratio/region_mean": 0.0007146268180804327, |
| "completions/clipped_ratio": 0.375, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13494.0, |
| "completions/mean_length": 13404.625, |
| "completions/mean_terminated_length": 11617.0, |
| "completions/min_length": 9977.0, |
| "completions/min_terminated_length": 9977.0, |
| "entropy": 0.43265021592378616, |
| "epoch": 0.019370042533624555, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.00584243331104517, |
| "learning_rate": 1e-05, |
| "loss": 0.0768, |
| "num_tokens": 15813009.0, |
| "reward": 0.125, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.125, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999116063117981, |
| "sampling/importance_sampling_ratio/min": 0.002006195019930601, |
| "sampling/sampling_logp_difference/max": 6.211515426635742, |
| "sampling/sampling_logp_difference/mean": 0.018601318821310997, |
| "step": 337 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13395.0, |
| "completions/max_terminated_length": 13395.0, |
| "completions/mean_length": 9461.375, |
| "completions/mean_terminated_length": 9461.375, |
| "completions/min_length": 6637.0, |
| "completions/min_terminated_length": 6637.0, |
| "entropy": 0.8805797770619392, |
| "epoch": 0.01942752040464421, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 15890124.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001945495605469, |
| "sampling/importance_sampling_ratio/min": 0.3986915946006775, |
| "sampling/sampling_logp_difference/max": 0.9195671081542969, |
| "sampling/sampling_logp_difference/mean": 0.02378488890826702, |
| "step": 338 |
| }, |
| { |
| "clip_ratio/high_max": 0.00014442485553445294, |
| "clip_ratio/high_mean": 0.00014442485553445294, |
| "clip_ratio/low_mean": 0.00019609275477705523, |
| "clip_ratio/low_min": 0.00019609275477705523, |
| "clip_ratio/region_mean": 0.0003405176103115082, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 10936.0, |
| "completions/mean_length": 9337.75, |
| "completions/mean_terminated_length": 8331.1435546875, |
| "completions/min_length": 6615.0, |
| "completions/min_terminated_length": 6615.0, |
| "entropy": 0.6978324055671692, |
| "epoch": 0.01948499827566387, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01227813120931387, |
| "learning_rate": 1e-05, |
| "loss": 0.1759, |
| "num_tokens": 15966842.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.8344930410385132, |
| "sampling/importance_sampling_ratio/mean": 1.0003163814544678, |
| "sampling/importance_sampling_ratio/min": 0.4923921227455139, |
| "sampling/sampling_logp_difference/max": 0.7084798812866211, |
| "sampling/sampling_logp_difference/mean": 0.02756013721227646, |
| "step": 339 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6524.0, |
| "completions/max_terminated_length": 6524.0, |
| "completions/mean_length": 4721.375, |
| "completions/mean_terminated_length": 4721.375, |
| "completions/min_length": 3315.0, |
| "completions/min_terminated_length": 3315.0, |
| "entropy": 0.32055202685296535, |
| "epoch": 0.019542476146683526, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16005621.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3077168464660645, |
| "sampling/importance_sampling_ratio/mean": 1.000072717666626, |
| "sampling/importance_sampling_ratio/min": 0.4791582226753235, |
| "sampling/sampling_logp_difference/max": 0.7357244491577148, |
| "sampling/sampling_logp_difference/mean": 0.0142374187707901, |
| "step": 340 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3782.0, |
| "completions/max_terminated_length": 3782.0, |
| "completions/mean_length": 2638.75, |
| "completions/mean_terminated_length": 2638.75, |
| "completions/min_length": 1449.0, |
| "completions/min_terminated_length": 1449.0, |
| "entropy": 0.32627006992697716, |
| "epoch": 0.019599954017703185, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16028171.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6159237623214722, |
| "sampling/importance_sampling_ratio/mean": 0.9995152354240417, |
| "sampling/importance_sampling_ratio/min": 0.6188905239105225, |
| "sampling/sampling_logp_difference/max": 0.4799067974090576, |
| "sampling/sampling_logp_difference/mean": 0.015043683350086212, |
| "step": 341 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00030060322387726046, |
| "clip_ratio/low_min": 0.00030060322387726046, |
| "clip_ratio/region_mean": 0.00030060322387726046, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11077.0, |
| "completions/max_terminated_length": 11077.0, |
| "completions/mean_length": 4994.375, |
| "completions/mean_terminated_length": 4994.375, |
| "completions/min_length": 1190.0, |
| "completions/min_terminated_length": 1190.0, |
| "entropy": 1.0550551414489746, |
| "epoch": 0.01965743188872284, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.017271967604756355, |
| "learning_rate": 1e-05, |
| "loss": 0.2909, |
| "num_tokens": 16069574.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.6729991436004639, |
| "sampling/importance_sampling_ratio/mean": 1.0001438856124878, |
| "sampling/importance_sampling_ratio/min": 0.4280719459056854, |
| "sampling/sampling_logp_difference/max": 0.8484640121459961, |
| "sampling/sampling_logp_difference/mean": 0.026202231645584106, |
| "step": 342 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14414.0, |
| "completions/mean_length": 12635.25, |
| "completions/mean_terminated_length": 12099.71484375, |
| "completions/min_length": 8481.0, |
| "completions/min_terminated_length": 8481.0, |
| "entropy": 0.6609185971319675, |
| "epoch": 0.0197149097597425, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16171688.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999833881855011, |
| "sampling/importance_sampling_ratio/min": 0.34503352642059326, |
| "sampling/sampling_logp_difference/max": 1.0641136169433594, |
| "sampling/sampling_logp_difference/mean": 0.0253476370126009, |
| "step": 343 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6682.0, |
| "completions/max_terminated_length": 6682.0, |
| "completions/mean_length": 3645.875, |
| "completions/mean_terminated_length": 3645.875, |
| "completions/min_length": 1887.0, |
| "completions/min_terminated_length": 1887.0, |
| "entropy": 0.39544339664280415, |
| "epoch": 0.019772387630762157, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16201879.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.9815831184387207, |
| "sampling/importance_sampling_ratio/mean": 1.0000919103622437, |
| "sampling/importance_sampling_ratio/min": 0.5613101124763489, |
| "sampling/sampling_logp_difference/max": 0.6838960647583008, |
| "sampling/sampling_logp_difference/mean": 0.016911331564188004, |
| "step": 344 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 14822.0, |
| "completions/mean_length": 14651.375, |
| "completions/mean_terminated_length": 12918.75, |
| "completions/min_length": 11623.0, |
| "completions/min_terminated_length": 11623.0, |
| "entropy": 0.42336586117744446, |
| "epoch": 0.019829865501781813, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16320738.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000054955482483, |
| "sampling/importance_sampling_ratio/min": 0.1662822961807251, |
| "sampling/sampling_logp_difference/max": 1.7940683364868164, |
| "sampling/sampling_logp_difference/mean": 0.019420776516199112, |
| "step": 345 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 11950.0, |
| "completions/max_terminated_length": 11950.0, |
| "completions/mean_length": 6997.125, |
| "completions/mean_terminated_length": 6997.125, |
| "completions/min_length": 4513.0, |
| "completions/min_terminated_length": 4513.0, |
| "entropy": 0.2710230275988579, |
| "epoch": 0.019887343372801472, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16377835.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7474852800369263, |
| "sampling/importance_sampling_ratio/mean": 1.0000462532043457, |
| "sampling/importance_sampling_ratio/min": 0.4111810028553009, |
| "sampling/sampling_logp_difference/max": 0.8887218236923218, |
| "sampling/sampling_logp_difference/mean": 0.012591187842190266, |
| "step": 346 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5051.0, |
| "completions/max_terminated_length": 5051.0, |
| "completions/mean_length": 2853.625, |
| "completions/mean_terminated_length": 2853.625, |
| "completions/min_length": 1698.0, |
| "completions/min_terminated_length": 1698.0, |
| "entropy": 0.4040024057030678, |
| "epoch": 0.019944821243821128, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16401672.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.735011339187622, |
| "sampling/importance_sampling_ratio/mean": 0.9998360872268677, |
| "sampling/importance_sampling_ratio/min": 0.49195700883865356, |
| "sampling/sampling_logp_difference/max": 0.7093639373779297, |
| "sampling/sampling_logp_difference/mean": 0.016577383503317833, |
| "step": 347 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2932.0, |
| "completions/max_terminated_length": 2932.0, |
| "completions/mean_length": 1817.0, |
| "completions/mean_terminated_length": 1817.0, |
| "completions/min_length": 759.0, |
| "completions/min_terminated_length": 759.0, |
| "entropy": 0.3871239461004734, |
| "epoch": 0.020002299114840787, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16417128.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5381892919540405, |
| "sampling/importance_sampling_ratio/mean": 1.0006383657455444, |
| "sampling/importance_sampling_ratio/min": 0.623935341835022, |
| "sampling/sampling_logp_difference/max": 0.4717085361480713, |
| "sampling/sampling_logp_difference/mean": 0.016869783401489258, |
| "step": 348 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5132.0, |
| "completions/max_terminated_length": 5132.0, |
| "completions/mean_length": 3433.0, |
| "completions/mean_terminated_length": 3433.0, |
| "completions/min_length": 2218.0, |
| "completions/min_terminated_length": 2218.0, |
| "entropy": 0.6093649119138718, |
| "epoch": 0.020059776985860443, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16445944.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.5278382301330566, |
| "sampling/importance_sampling_ratio/mean": 0.9998639822006226, |
| "sampling/importance_sampling_ratio/min": 0.5801088809967041, |
| "sampling/sampling_logp_difference/max": 0.5445394515991211, |
| "sampling/sampling_logp_difference/mean": 0.023235652595758438, |
| "step": 349 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 13434.0, |
| "completions/max_terminated_length": 13434.0, |
| "completions/mean_length": 9765.5, |
| "completions/mean_terminated_length": 9765.5, |
| "completions/min_length": 5986.0, |
| "completions/min_terminated_length": 5986.0, |
| "entropy": 0.9009614214301109, |
| "epoch": 0.020117254856880103, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16524996.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9995908737182617, |
| "sampling/importance_sampling_ratio/min": 0.3160809278488159, |
| "sampling/sampling_logp_difference/max": 1.1751902103424072, |
| "sampling/sampling_logp_difference/mean": 0.02548110857605934, |
| "step": 350 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9958.0, |
| "completions/max_terminated_length": 9958.0, |
| "completions/mean_length": 5576.125, |
| "completions/mean_terminated_length": 5576.125, |
| "completions/min_length": 1937.0, |
| "completions/min_terminated_length": 1937.0, |
| "entropy": 1.234039157629013, |
| "epoch": 0.02017473272789976, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16571245.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6989067792892456, |
| "sampling/importance_sampling_ratio/mean": 0.9999986886978149, |
| "sampling/importance_sampling_ratio/min": 0.5405990481376648, |
| "sampling/sampling_logp_difference/max": 0.6150774955749512, |
| "sampling/sampling_logp_difference/mean": 0.030534079298377037, |
| "step": 351 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012208950647618622, |
| "clip_ratio/high_mean": 0.00012208950647618622, |
| "clip_ratio/low_mean": 6.016847328282893e-05, |
| "clip_ratio/low_min": 6.016847328282893e-05, |
| "clip_ratio/region_mean": 0.00018225797975901514, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5135.0, |
| "completions/max_terminated_length": 5135.0, |
| "completions/mean_length": 3478.125, |
| "completions/mean_terminated_length": 3478.125, |
| "completions/min_length": 2563.0, |
| "completions/min_terminated_length": 2563.0, |
| "entropy": 0.3595081530511379, |
| "epoch": 0.020232210598919415, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.009329551830887794, |
| "learning_rate": 1e-05, |
| "loss": 0.0687, |
| "num_tokens": 16600134.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.5045642852783203, |
| "sampling/importance_sampling_ratio/mean": 1.0004687309265137, |
| "sampling/importance_sampling_ratio/min": 0.5318603515625, |
| "sampling/sampling_logp_difference/max": 0.6313743591308594, |
| "sampling/sampling_logp_difference/mean": 0.014808500185608864, |
| "step": 352 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00010229132749373093, |
| "clip_ratio/low_min": 0.00010229132749373093, |
| "clip_ratio/region_mean": 0.00010229132749373093, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2319.0, |
| "completions/max_terminated_length": 2319.0, |
| "completions/mean_length": 1595.875, |
| "completions/mean_terminated_length": 1595.875, |
| "completions/min_length": 865.0, |
| "completions/min_terminated_length": 865.0, |
| "entropy": 0.35693978890776634, |
| "epoch": 0.020289688469939074, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024179993197321892, |
| "learning_rate": 1e-05, |
| "loss": -0.1834, |
| "num_tokens": 16613805.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4377808570861816, |
| "sampling/importance_sampling_ratio/mean": 1.0000147819519043, |
| "sampling/importance_sampling_ratio/min": 0.57708340883255, |
| "sampling/sampling_logp_difference/max": 0.5497684478759766, |
| "sampling/sampling_logp_difference/mean": 0.015388967469334602, |
| "step": 353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 1.0, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 0.0, |
| "completions/mean_length": 16384.0, |
| "completions/mean_terminated_length": 0.0, |
| "completions/min_length": 16384.0, |
| "completions/min_terminated_length": 0.0, |
| "entropy": 0.66748046875, |
| "epoch": 0.02034716634095873, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16746693.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000501871109009, |
| "sampling/importance_sampling_ratio/min": 0.17032411694526672, |
| "sampling/sampling_logp_difference/max": 1.7700520753860474, |
| "sampling/sampling_logp_difference/mean": 0.02706432342529297, |
| "step": 354 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3290.0, |
| "completions/max_terminated_length": 3290.0, |
| "completions/mean_length": 1433.75, |
| "completions/mean_terminated_length": 1433.75, |
| "completions/min_length": 506.0, |
| "completions/min_terminated_length": 506.0, |
| "entropy": 0.4565188344568014, |
| "epoch": 0.02040464421197839, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16759131.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.3809913396835327, |
| "sampling/importance_sampling_ratio/mean": 0.999972939491272, |
| "sampling/importance_sampling_ratio/min": 0.61844402551651, |
| "sampling/sampling_logp_difference/max": 0.480548620223999, |
| "sampling/sampling_logp_difference/mean": 0.021480221301317215, |
| "step": 355 |
| }, |
| { |
| "clip_ratio/high_max": 3.810596899711527e-05, |
| "clip_ratio/high_mean": 3.810596899711527e-05, |
| "clip_ratio/low_mean": 0.00023517667432315648, |
| "clip_ratio/low_min": 0.00023517667432315648, |
| "clip_ratio/region_mean": 0.00027328264332027175, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 11950.0, |
| "completions/mean_length": 10748.875, |
| "completions/mean_terminated_length": 8870.5, |
| "completions/min_length": 5878.0, |
| "completions/min_terminated_length": 5878.0, |
| "entropy": 0.3735651969909668, |
| "epoch": 0.020462122082998045, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02715384028851986, |
| "learning_rate": 1e-05, |
| "loss": 0.2014, |
| "num_tokens": 16846426.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001177787780762, |
| "sampling/importance_sampling_ratio/min": 0.3348197340965271, |
| "sampling/sampling_logp_difference/max": 1.094162940979004, |
| "sampling/sampling_logp_difference/mean": 0.015553958714008331, |
| "step": 356 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12477.0, |
| "completions/max_terminated_length": 12477.0, |
| "completions/mean_length": 7907.25, |
| "completions/mean_terminated_length": 7907.25, |
| "completions/min_length": 5728.0, |
| "completions/min_terminated_length": 5728.0, |
| "entropy": 0.6937508285045624, |
| "epoch": 0.020519599954017705, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 16910732.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001392364501953, |
| "sampling/importance_sampling_ratio/min": 0.4836191236972809, |
| "sampling/sampling_logp_difference/max": 0.8179230690002441, |
| "sampling/sampling_logp_difference/mean": 0.027213335037231445, |
| "step": 357 |
| }, |
| { |
| "clip_ratio/high_max": 9.627788131183479e-05, |
| "clip_ratio/high_mean": 9.627788131183479e-05, |
| "clip_ratio/low_mean": 0.0003370559716131538, |
| "clip_ratio/low_min": 0.0003370559716131538, |
| "clip_ratio/region_mean": 0.0004333338529249886, |
| "completions/clipped_ratio": 0.25, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 13925.0, |
| "completions/mean_length": 11331.5, |
| "completions/mean_terminated_length": 9647.333984375, |
| "completions/min_length": 5608.0, |
| "completions/min_terminated_length": 5608.0, |
| "entropy": 0.3750923369079828, |
| "epoch": 0.02057707782503736, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.023595118895173073, |
| "learning_rate": 1e-05, |
| "loss": 0.2705, |
| "num_tokens": 17002504.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.999678909778595, |
| "sampling/importance_sampling_ratio/min": 0.310112327337265, |
| "sampling/sampling_logp_difference/max": 1.170820713043213, |
| "sampling/sampling_logp_difference/mean": 0.017307985574007034, |
| "step": 358 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9512.0, |
| "completions/max_terminated_length": 9512.0, |
| "completions/mean_length": 4008.875, |
| "completions/mean_terminated_length": 4008.875, |
| "completions/min_length": 1897.0, |
| "completions/min_terminated_length": 1897.0, |
| "entropy": 0.603480126708746, |
| "epoch": 0.020634555696057016, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17035511.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7548553943634033, |
| "sampling/importance_sampling_ratio/mean": 0.9997280240058899, |
| "sampling/importance_sampling_ratio/min": 0.48254984617233276, |
| "sampling/sampling_logp_difference/max": 0.7286710739135742, |
| "sampling/sampling_logp_difference/mean": 0.022598395124077797, |
| "step": 359 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 15712.0, |
| "completions/max_terminated_length": 15712.0, |
| "completions/mean_length": 12256.25, |
| "completions/mean_terminated_length": 12256.25, |
| "completions/min_length": 7515.0, |
| "completions/min_terminated_length": 7515.0, |
| "entropy": 0.9838720113039017, |
| "epoch": 0.020692033567076676, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17135345.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.00010347366333, |
| "sampling/importance_sampling_ratio/min": 0.2992424964904785, |
| "sampling/sampling_logp_difference/max": 1.2065010070800781, |
| "sampling/sampling_logp_difference/mean": 0.0314478874206543, |
| "step": 360 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3067.0, |
| "completions/max_terminated_length": 3067.0, |
| "completions/mean_length": 2019.5, |
| "completions/mean_terminated_length": 2019.5, |
| "completions/min_length": 865.0, |
| "completions/min_terminated_length": 865.0, |
| "entropy": 0.4162302501499653, |
| "epoch": 0.020749511438096332, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17152701.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.4723769426345825, |
| "sampling/importance_sampling_ratio/mean": 0.9998703598976135, |
| "sampling/importance_sampling_ratio/min": 0.6035754084587097, |
| "sampling/sampling_logp_difference/max": 0.5048842430114746, |
| "sampling/sampling_logp_difference/mean": 0.018714936450123787, |
| "step": 361 |
| }, |
| { |
| "clip_ratio/high_max": 5.953218715148978e-05, |
| "clip_ratio/high_mean": 5.953218715148978e-05, |
| "clip_ratio/low_mean": 0.0005276044685160741, |
| "clip_ratio/low_min": 0.0005276044685160741, |
| "clip_ratio/region_mean": 0.0005871366556675639, |
| "completions/clipped_ratio": 0.5, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15048.0, |
| "completions/mean_length": 12956.375, |
| "completions/mean_terminated_length": 9528.75, |
| "completions/min_length": 6243.0, |
| "completions/min_terminated_length": 6243.0, |
| "entropy": 0.44832662865519524, |
| "epoch": 0.02080698930911599, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01465114951133728, |
| "learning_rate": 1e-05, |
| "loss": 0.2945, |
| "num_tokens": 17258872.0, |
| "reward": 0.375, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.375, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999077916145325, |
| "sampling/importance_sampling_ratio/min": 0.3361400067806244, |
| "sampling/sampling_logp_difference/max": 1.154249668121338, |
| "sampling/sampling_logp_difference/mean": 0.01987970992922783, |
| "step": 362 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10306.0, |
| "completions/max_terminated_length": 10306.0, |
| "completions/mean_length": 4658.25, |
| "completions/mean_terminated_length": 4658.25, |
| "completions/min_length": 1282.0, |
| "completions/min_terminated_length": 1282.0, |
| "entropy": 0.6721976362168789, |
| "epoch": 0.020864467180135647, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17297490.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7022991180419922, |
| "sampling/importance_sampling_ratio/mean": 1.0000230073928833, |
| "sampling/importance_sampling_ratio/min": 0.0958167016506195, |
| "sampling/sampling_logp_difference/max": 2.34531831741333, |
| "sampling/sampling_logp_difference/mean": 0.02582489140331745, |
| "step": 363 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6522.0, |
| "completions/max_terminated_length": 6522.0, |
| "completions/mean_length": 2912.0, |
| "completions/mean_terminated_length": 2912.0, |
| "completions/min_length": 2044.0, |
| "completions/min_terminated_length": 2044.0, |
| "entropy": 0.22699221409857273, |
| "epoch": 0.020921945051155307, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17321722.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6960196495056152, |
| "sampling/importance_sampling_ratio/mean": 1.0000361204147339, |
| "sampling/importance_sampling_ratio/min": 0.4738141596317291, |
| "sampling/sampling_logp_difference/max": 0.7469401359558105, |
| "sampling/sampling_logp_difference/mean": 0.010224299505352974, |
| "step": 364 |
| }, |
| { |
| "clip_ratio/high_max": 7.319757969526108e-05, |
| "clip_ratio/high_mean": 7.319757969526108e-05, |
| "clip_ratio/low_mean": 0.00011551292845979333, |
| "clip_ratio/low_min": 0.00011551292845979333, |
| "clip_ratio/region_mean": 0.0001887105081550544, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 8141.0, |
| "completions/mean_length": 7507.125, |
| "completions/mean_terminated_length": 6239.00048828125, |
| "completions/min_length": 3957.0, |
| "completions/min_terminated_length": 3957.0, |
| "entropy": 0.3593865130096674, |
| "epoch": 0.020979422922174962, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010079027153551579, |
| "learning_rate": 1e-05, |
| "loss": 0.1919, |
| "num_tokens": 17383611.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.6150532960891724, |
| "sampling/importance_sampling_ratio/mean": 0.9999769926071167, |
| "sampling/importance_sampling_ratio/min": 0.15059253573417664, |
| "sampling/sampling_logp_difference/max": 1.8931775093078613, |
| "sampling/sampling_logp_difference/mean": 0.016810156404972076, |
| "step": 365 |
| }, |
| { |
| "clip_ratio/high_max": 7.529172580689192e-05, |
| "clip_ratio/high_mean": 7.529172580689192e-05, |
| "clip_ratio/low_mean": 0.0001947502387338318, |
| "clip_ratio/low_min": 0.0001947502387338318, |
| "clip_ratio/region_mean": 0.0002700419645407237, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12793.0, |
| "completions/mean_length": 10771.375, |
| "completions/mean_terminated_length": 9969.572265625, |
| "completions/min_length": 6930.0, |
| "completions/min_terminated_length": 6930.0, |
| "entropy": 0.7980213016271591, |
| "epoch": 0.02103690079319462, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.010477994568645954, |
| "learning_rate": 1e-05, |
| "loss": 0.1912, |
| "num_tokens": 17470710.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.9356058835983276, |
| "sampling/importance_sampling_ratio/mean": 1.0001732110977173, |
| "sampling/importance_sampling_ratio/min": 0.43703436851501465, |
| "sampling/sampling_logp_difference/max": 0.8277435302734375, |
| "sampling/sampling_logp_difference/mean": 0.025370774790644646, |
| "step": 366 |
| }, |
| { |
| "clip_ratio/high_max": 6.0065296565881e-05, |
| "clip_ratio/high_mean": 6.0065296565881e-05, |
| "clip_ratio/low_mean": 0.0005909008614253253, |
| "clip_ratio/low_min": 0.0005909008614253253, |
| "clip_ratio/region_mean": 0.0006509661579912063, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15605.0, |
| "completions/mean_length": 9916.125, |
| "completions/mean_terminated_length": 8992.1435546875, |
| "completions/min_length": 4208.0, |
| "completions/min_terminated_length": 4208.0, |
| "entropy": 0.45364048704504967, |
| "epoch": 0.021094378664214278, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.008153869770467281, |
| "learning_rate": 1e-05, |
| "loss": -0.0246, |
| "num_tokens": 17551735.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9997556209564209, |
| "sampling/importance_sampling_ratio/min": 0.268157958984375, |
| "sampling/sampling_logp_difference/max": 1.3161790370941162, |
| "sampling/sampling_logp_difference/mean": 0.019931312650442123, |
| "step": 367 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 4067.0, |
| "completions/max_terminated_length": 4067.0, |
| "completions/mean_length": 2725.0, |
| "completions/mean_terminated_length": 2725.0, |
| "completions/min_length": 1576.0, |
| "completions/min_terminated_length": 1576.0, |
| "entropy": 0.5497967749834061, |
| "epoch": 0.021151856535233934, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17574487.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6456069946289062, |
| "sampling/importance_sampling_ratio/mean": 1.00068199634552, |
| "sampling/importance_sampling_ratio/min": 0.609940230846405, |
| "sampling/sampling_logp_difference/max": 0.4981093406677246, |
| "sampling/sampling_logp_difference/mean": 0.021459870040416718, |
| "step": 368 |
| }, |
| { |
| "clip_ratio/high_max": 9.761245200934354e-05, |
| "clip_ratio/high_mean": 9.761245200934354e-05, |
| "clip_ratio/low_mean": 0.00023482833057641983, |
| "clip_ratio/low_min": 0.00023482833057641983, |
| "clip_ratio/region_mean": 0.00033244078258576337, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 15704.0, |
| "completions/mean_length": 11187.5, |
| "completions/mean_terminated_length": 10445.1435546875, |
| "completions/min_length": 4421.0, |
| "completions/min_terminated_length": 4421.0, |
| "entropy": 0.37645208090543747, |
| "epoch": 0.021209334406253593, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.03603425621986389, |
| "learning_rate": 1e-05, |
| "loss": 0.2343, |
| "num_tokens": 17665315.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999097585678101, |
| "sampling/importance_sampling_ratio/min": 0.16250598430633545, |
| "sampling/sampling_logp_difference/max": 1.8170404434204102, |
| "sampling/sampling_logp_difference/mean": 0.018346307799220085, |
| "step": 369 |
| }, |
| { |
| "clip_ratio/high_max": 1.8873623048420995e-05, |
| "clip_ratio/high_mean": 1.8873623048420995e-05, |
| "clip_ratio/low_mean": 0.00012164050713181496, |
| "clip_ratio/low_min": 0.00012164050713181496, |
| "clip_ratio/region_mean": 0.00014051413018023595, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 12125.0, |
| "completions/max_terminated_length": 12125.0, |
| "completions/mean_length": 5782.25, |
| "completions/mean_terminated_length": 5782.25, |
| "completions/min_length": 1970.0, |
| "completions/min_terminated_length": 1970.0, |
| "entropy": 0.6423214338719845, |
| "epoch": 0.02126681227727325, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.02861444652080536, |
| "learning_rate": 1e-05, |
| "loss": -0.0707, |
| "num_tokens": 17712965.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.8723466396331787, |
| "sampling/importance_sampling_ratio/mean": 0.9999862909317017, |
| "sampling/importance_sampling_ratio/min": 0.4887623190879822, |
| "sampling/sampling_logp_difference/max": 0.715878963470459, |
| "sampling/sampling_logp_difference/mean": 0.018169758841395378, |
| "step": 370 |
| }, |
| { |
| "clip_ratio/high_max": 0.0002698010503081605, |
| "clip_ratio/high_mean": 0.0002698010503081605, |
| "clip_ratio/low_mean": 0.00021440846467157826, |
| "clip_ratio/low_min": 0.00021440846467157826, |
| "clip_ratio/region_mean": 0.00048420951497973874, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2882.0, |
| "completions/max_terminated_length": 2882.0, |
| "completions/mean_length": 2059.5, |
| "completions/mean_terminated_length": 2059.5, |
| "completions/min_length": 864.0, |
| "completions/min_terminated_length": 864.0, |
| "entropy": 0.5068723000586033, |
| "epoch": 0.02132429014829291, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.08042344450950623, |
| "learning_rate": 1e-05, |
| "loss": 0.0347, |
| "num_tokens": 17730601.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.5238127708435059, |
| "sampling/importance_sampling_ratio/mean": 0.9996336698532104, |
| "sampling/importance_sampling_ratio/min": 0.5397042036056519, |
| "sampling/sampling_logp_difference/max": 0.6167340278625488, |
| "sampling/sampling_logp_difference/mean": 0.021951928734779358, |
| "step": 371 |
| }, |
| { |
| "clip_ratio/high_max": 1.83769479917828e-05, |
| "clip_ratio/high_mean": 1.83769479917828e-05, |
| "clip_ratio/low_mean": 0.00040829650970408693, |
| "clip_ratio/low_min": 0.00040829650970408693, |
| "clip_ratio/region_mean": 0.00042667345769586973, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9130.0, |
| "completions/max_terminated_length": 9130.0, |
| "completions/mean_length": 6032.375, |
| "completions/mean_terminated_length": 6032.375, |
| "completions/min_length": 3842.0, |
| "completions/min_terminated_length": 3842.0, |
| "entropy": 0.3433102909475565, |
| "epoch": 0.021381768019312564, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.022114230319857597, |
| "learning_rate": 1e-05, |
| "loss": 0.1585, |
| "num_tokens": 17780132.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.5154314041137695, |
| "sampling/importance_sampling_ratio/mean": 1.0000022649765015, |
| "sampling/importance_sampling_ratio/min": 0.5724499821662903, |
| "sampling/sampling_logp_difference/max": 0.5578298568725586, |
| "sampling/sampling_logp_difference/mean": 0.014348246157169342, |
| "step": 372 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 8.632597018731758e-05, |
| "clip_ratio/low_min": 8.632597018731758e-05, |
| "clip_ratio/region_mean": 8.632597018731758e-05, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 3596.0, |
| "completions/max_terminated_length": 3596.0, |
| "completions/mean_length": 2008.5, |
| "completions/mean_terminated_length": 2008.5, |
| "completions/min_length": 964.0, |
| "completions/min_terminated_length": 964.0, |
| "entropy": 0.3825047891587019, |
| "epoch": 0.021439245890332224, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.024160167202353477, |
| "learning_rate": 1e-05, |
| "loss": -0.2159, |
| "num_tokens": 17797496.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4042750597000122, |
| "sampling/importance_sampling_ratio/mean": 1.0002102851867676, |
| "sampling/importance_sampling_ratio/min": 0.7001754641532898, |
| "sampling/sampling_logp_difference/max": 0.35642433166503906, |
| "sampling/sampling_logp_difference/mean": 0.013086330145597458, |
| "step": 373 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10115.0, |
| "completions/max_terminated_length": 10115.0, |
| "completions/mean_length": 5735.125, |
| "completions/mean_terminated_length": 5735.125, |
| "completions/min_length": 3126.0, |
| "completions/min_terminated_length": 3126.0, |
| "entropy": 0.9846834391355515, |
| "epoch": 0.02149672376135188, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17844697.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001513957977295, |
| "sampling/importance_sampling_ratio/min": 0.41287025809288025, |
| "sampling/sampling_logp_difference/max": 0.8846219182014465, |
| "sampling/sampling_logp_difference/mean": 0.02507157251238823, |
| "step": 374 |
| }, |
| { |
| "clip_ratio/high_max": 0.00012527878880064236, |
| "clip_ratio/high_mean": 0.00012527878880064236, |
| "clip_ratio/low_mean": 0.00017507003212813288, |
| "clip_ratio/low_min": 0.00017507003212813288, |
| "clip_ratio/region_mean": 0.00030034882092877524, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10439.0, |
| "completions/max_terminated_length": 10439.0, |
| "completions/mean_length": 6156.875, |
| "completions/mean_terminated_length": 6156.875, |
| "completions/min_length": 2142.0, |
| "completions/min_terminated_length": 2142.0, |
| "entropy": 0.846487246453762, |
| "epoch": 0.021554201632371536, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.01263112761080265, |
| "learning_rate": 1e-05, |
| "loss": -0.2307, |
| "num_tokens": 17895160.0, |
| "reward": 0.875, |
| "reward_std": 0.3535533845424652, |
| "rewards/accuracy_reward/mean": 0.875, |
| "rewards/accuracy_reward/std": 0.3535533845424652, |
| "sampling/importance_sampling_ratio/max": 1.6920719146728516, |
| "sampling/importance_sampling_ratio/mean": 0.9999474287033081, |
| "sampling/importance_sampling_ratio/min": 0.2458421289920807, |
| "sampling/sampling_logp_difference/max": 1.4030656814575195, |
| "sampling/sampling_logp_difference/mean": 0.028860243037343025, |
| "step": 375 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 7747.0, |
| "completions/max_terminated_length": 7747.0, |
| "completions/mean_length": 4877.375, |
| "completions/mean_terminated_length": 4877.375, |
| "completions/min_length": 2300.0, |
| "completions/min_terminated_length": 2300.0, |
| "entropy": 0.3635020200163126, |
| "epoch": 0.021611679503391195, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 17935579.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.6226881742477417, |
| "sampling/importance_sampling_ratio/mean": 0.9995774626731873, |
| "sampling/importance_sampling_ratio/min": 0.5516060590744019, |
| "sampling/sampling_logp_difference/max": 0.5949211120605469, |
| "sampling/sampling_logp_difference/mean": 0.01612044870853424, |
| "step": 376 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.00013702968863071874, |
| "clip_ratio/low_min": 0.00013702968863071874, |
| "clip_ratio/region_mean": 0.00013702968863071874, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 6776.0, |
| "completions/max_terminated_length": 6776.0, |
| "completions/mean_length": 3584.125, |
| "completions/mean_terminated_length": 3584.125, |
| "completions/min_length": 1255.0, |
| "completions/min_terminated_length": 1255.0, |
| "entropy": 1.4472458511590958, |
| "epoch": 0.02166915737441085, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.09286519885063171, |
| "learning_rate": 1e-05, |
| "loss": -0.1914, |
| "num_tokens": 17965828.0, |
| "reward": 0.25, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.25, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 1.4383729696273804, |
| "sampling/importance_sampling_ratio/mean": 0.9994940161705017, |
| "sampling/importance_sampling_ratio/min": 0.530176043510437, |
| "sampling/sampling_logp_difference/max": 0.6345462799072266, |
| "sampling/sampling_logp_difference/mean": 0.03094877488911152, |
| "step": 377 |
| }, |
| { |
| "clip_ratio/high_max": 8.652422911836766e-05, |
| "clip_ratio/high_mean": 8.652422911836766e-05, |
| "clip_ratio/low_mean": 8.914233876566868e-05, |
| "clip_ratio/low_min": 8.914233876566868e-05, |
| "clip_ratio/region_mean": 0.00017566656788403634, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8465.0, |
| "completions/max_terminated_length": 8465.0, |
| "completions/mean_length": 4396.375, |
| "completions/mean_terminated_length": 4396.375, |
| "completions/min_length": 2541.0, |
| "completions/min_terminated_length": 2541.0, |
| "entropy": 0.845998540520668, |
| "epoch": 0.02172663524543051, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.07467586547136307, |
| "learning_rate": 1e-05, |
| "loss": 0.2373, |
| "num_tokens": 18002471.0, |
| "reward": 0.75, |
| "reward_std": 0.4629100561141968, |
| "rewards/accuracy_reward/mean": 0.75, |
| "rewards/accuracy_reward/std": 0.4629100561141968, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000317931175232, |
| "sampling/importance_sampling_ratio/min": 0.4164700210094452, |
| "sampling/sampling_logp_difference/max": 0.9122521877288818, |
| "sampling/sampling_logp_difference/mean": 0.022327328100800514, |
| "step": 378 |
| }, |
| { |
| "clip_ratio/high_max": 7.681223723920994e-05, |
| "clip_ratio/high_mean": 7.681223723920994e-05, |
| "clip_ratio/low_mean": 0.0002786066252156161, |
| "clip_ratio/low_min": 0.0002786066252156161, |
| "clip_ratio/region_mean": 0.000355418862454826, |
| "completions/clipped_ratio": 0.125, |
| "completions/max_length": 16384.0, |
| "completions/max_terminated_length": 12453.0, |
| "completions/mean_length": 10504.0, |
| "completions/mean_terminated_length": 9664.0, |
| "completions/min_length": 6857.0, |
| "completions/min_terminated_length": 6857.0, |
| "entropy": 0.43550832010805607, |
| "epoch": 0.021784113116450166, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.007709996774792671, |
| "learning_rate": 1e-05, |
| "loss": 0.2212, |
| "num_tokens": 18087655.0, |
| "reward": 0.625, |
| "reward_std": 0.5175491571426392, |
| "rewards/accuracy_reward/mean": 0.625, |
| "rewards/accuracy_reward/std": 0.5175492167472839, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 1.000183343887329, |
| "sampling/importance_sampling_ratio/min": 0.5077204704284668, |
| "sampling/sampling_logp_difference/max": 0.8771915435791016, |
| "sampling/sampling_logp_difference/mean": 0.018277566879987717, |
| "step": 379 |
| }, |
| { |
| "clip_ratio/high_max": 4.323030952946283e-05, |
| "clip_ratio/high_mean": 4.323030952946283e-05, |
| "clip_ratio/low_mean": 0.0002932506122306222, |
| "clip_ratio/low_min": 0.0002932506122306222, |
| "clip_ratio/region_mean": 0.00033648092176008504, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 9631.0, |
| "completions/max_terminated_length": 9631.0, |
| "completions/mean_length": 7189.0, |
| "completions/mean_terminated_length": 7189.0, |
| "completions/min_length": 4571.0, |
| "completions/min_terminated_length": 4571.0, |
| "entropy": 0.8732515946030617, |
| "epoch": 0.021841590987469826, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 0.021911537274718285, |
| "learning_rate": 1e-05, |
| "loss": -0.005, |
| "num_tokens": 18146159.0, |
| "reward": 0.5, |
| "reward_std": 0.5345224738121033, |
| "rewards/accuracy_reward/mean": 0.5, |
| "rewards/accuracy_reward/std": 0.5345224738121033, |
| "sampling/importance_sampling_ratio/max": 1.6794453859329224, |
| "sampling/importance_sampling_ratio/mean": 0.9997799396514893, |
| "sampling/importance_sampling_ratio/min": 0.287708044052124, |
| "sampling/sampling_logp_difference/max": 1.2458090782165527, |
| "sampling/sampling_logp_difference/mean": 0.02427300438284874, |
| "step": 380 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 8625.0, |
| "completions/max_terminated_length": 8625.0, |
| "completions/mean_length": 5537.0, |
| "completions/mean_terminated_length": 5537.0, |
| "completions/min_length": 3167.0, |
| "completions/min_terminated_length": 3167.0, |
| "entropy": 0.8039069324731827, |
| "epoch": 0.02189906885848948, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 18191703.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998453855514526, |
| "sampling/importance_sampling_ratio/min": 0.5594327449798584, |
| "sampling/sampling_logp_difference/max": 0.7885527610778809, |
| "sampling/sampling_logp_difference/mean": 0.030826354399323463, |
| "step": 381 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 2796.0, |
| "completions/max_terminated_length": 2796.0, |
| "completions/mean_length": 2149.625, |
| "completions/mean_terminated_length": 2149.625, |
| "completions/min_length": 1581.0, |
| "completions/min_terminated_length": 1581.0, |
| "entropy": 0.14457772858440876, |
| "epoch": 0.021956546729509138, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 18210036.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.2777059078216553, |
| "sampling/importance_sampling_ratio/mean": 0.9998260140419006, |
| "sampling/importance_sampling_ratio/min": 0.6714066863059998, |
| "sampling/sampling_logp_difference/max": 0.3983802795410156, |
| "sampling/sampling_logp_difference/mean": 0.006943520158529282, |
| "step": 382 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 10632.0, |
| "completions/max_terminated_length": 10632.0, |
| "completions/mean_length": 9324.0, |
| "completions/mean_terminated_length": 9324.0, |
| "completions/min_length": 7091.0, |
| "completions/min_terminated_length": 7091.0, |
| "entropy": 1.1208415552973747, |
| "epoch": 0.022014024600528797, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 18285828.0, |
| "reward": 0.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 0.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 2.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999837875366211, |
| "sampling/importance_sampling_ratio/min": 0.19099198281764984, |
| "sampling/sampling_logp_difference/max": 1.6555237770080566, |
| "sampling/sampling_logp_difference/mean": 0.034007880836725235, |
| "step": 383 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 5809.0, |
| "completions/max_terminated_length": 5809.0, |
| "completions/mean_length": 3529.25, |
| "completions/mean_terminated_length": 3529.25, |
| "completions/min_length": 2132.0, |
| "completions/min_terminated_length": 2132.0, |
| "entropy": 0.365608349442482, |
| "epoch": 0.022071502471548453, |
| "frac_reward_zero_std": 1.0, |
| "grad_norm": 0.0, |
| "learning_rate": 1e-05, |
| "loss": 0.0, |
| "num_tokens": 18314958.0, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/accuracy_reward/mean": 1.0, |
| "rewards/accuracy_reward/std": 0.0, |
| "sampling/importance_sampling_ratio/max": 1.7880229949951172, |
| "sampling/importance_sampling_ratio/mean": 0.9999485015869141, |
| "sampling/importance_sampling_ratio/min": 0.6207014322280884, |
| "sampling/sampling_logp_difference/max": 0.5811104774475098, |
| "sampling/sampling_logp_difference/mean": 0.012056337669491768, |
| "step": 384 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 1024, |
| "num_input_tokens_seen": 18314958, |
| "num_train_epochs": 1, |
| "save_steps": 64, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|