| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.2857142857142857, |
| "eval_steps": 500, |
| "global_step": 100, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.1, |
| "completions/max_terminated_length": 16.1, |
| "completions/mean_length": 15.43125, |
| "completions/mean_terminated_length": 15.43125, |
| "completions/min_length": 12.8, |
| "completions/min_terminated_length": 12.8, |
| "entropy": 0.11648641154170036, |
| "epoch": 0.02857142857142857, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 3.766094207763672, |
| "learning_rate": 1.955e-06, |
| "loss": -0.03027614951133728, |
| "num_tokens": 117861.0, |
| "reward": 0.42463125213980674, |
| "reward_std": 0.4041272960603237, |
| "rewards/<lambda>/mean": 0.42463125213980674, |
| "rewards/<lambda>/std": 0.4041273109614849, |
| "step": 10, |
| "step_time": 8.86154415559722 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 39.1, |
| "completions/max_terminated_length": 39.1, |
| "completions/mean_length": 15.76875, |
| "completions/mean_terminated_length": 15.76875, |
| "completions/min_length": 12.0, |
| "completions/min_terminated_length": 12.0, |
| "entropy": 0.14287759065628053, |
| "epoch": 0.05714285714285714, |
| "frac_reward_zero_std": 0.3, |
| "grad_norm": 4.652894020080566, |
| "learning_rate": 1.905e-06, |
| "loss": -0.007219273597002029, |
| "num_tokens": 211120.0, |
| "reward": 0.16723437011241912, |
| "reward_std": 0.23346474170684814, |
| "rewards/<lambda>/mean": 0.16723437011241912, |
| "rewards/<lambda>/std": 0.233464752137661, |
| "step": 20, |
| "step_time": 8.861151697207243 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.6, |
| "completions/max_terminated_length": 16.6, |
| "completions/mean_length": 15.18125, |
| "completions/mean_terminated_length": 15.18125, |
| "completions/min_length": 12.6, |
| "completions/min_terminated_length": 12.6, |
| "entropy": 0.10504549741744995, |
| "epoch": 0.08571428571428572, |
| "frac_reward_zero_std": 0.1, |
| "grad_norm": 2.999187707901001, |
| "learning_rate": 1.8549999999999998e-06, |
| "loss": -0.01587933599948883, |
| "num_tokens": 315101.0, |
| "reward": 0.5093562602996826, |
| "reward_std": 0.2611938640475273, |
| "rewards/<lambda>/mean": 0.5093562602996826, |
| "rewards/<lambda>/std": 0.26119386702775954, |
| "step": 30, |
| "step_time": 8.02011652730871 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.83125, |
| "completions/mean_terminated_length": 15.83125, |
| "completions/min_length": 14.6, |
| "completions/min_terminated_length": 14.6, |
| "entropy": 0.07781314179301262, |
| "epoch": 0.11428571428571428, |
| "frac_reward_zero_std": 0.0, |
| "grad_norm": 2.083005666732788, |
| "learning_rate": 1.8049999999999999e-06, |
| "loss": -0.0016902854666113853, |
| "num_tokens": 431602.0, |
| "reward": 0.7693868696689605, |
| "reward_std": 0.3072405755519867, |
| "rewards/<lambda>/mean": 0.7693868696689605, |
| "rewards/<lambda>/std": 0.3072405830025673, |
| "step": 40, |
| "step_time": 8.891476768592838 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.36875, |
| "completions/mean_terminated_length": 15.36875, |
| "completions/min_length": 13.9, |
| "completions/min_terminated_length": 13.9, |
| "entropy": 0.09770065676420928, |
| "epoch": 0.14285714285714285, |
| "frac_reward_zero_std": 0.2, |
| "grad_norm": 0.0, |
| "learning_rate": 1.7549999999999997e-06, |
| "loss": -0.004019740223884583, |
| "num_tokens": 525533.0, |
| "reward": 0.5823375027626753, |
| "reward_std": 0.2531014457345009, |
| "rewards/<lambda>/mean": 0.5823375027626753, |
| "rewards/<lambda>/std": 0.25310145914554594, |
| "step": 50, |
| "step_time": 7.289726911496837 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.8875, |
| "completions/mean_terminated_length": 15.8875, |
| "completions/min_length": 15.5, |
| "completions/min_terminated_length": 15.5, |
| "entropy": 0.058196247555315495, |
| "epoch": 0.17142857142857143, |
| "frac_reward_zero_std": 0.2, |
| "grad_norm": 3.2708792686462402, |
| "learning_rate": 1.705e-06, |
| "loss": 0.0007023245096206665, |
| "num_tokens": 633339.0, |
| "reward": 0.8079562425613404, |
| "reward_std": 0.26226545721292494, |
| "rewards/<lambda>/mean": 0.8079562425613404, |
| "rewards/<lambda>/std": 0.26226547062397004, |
| "step": 60, |
| "step_time": 8.16725606439868 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.86875, |
| "completions/mean_terminated_length": 15.86875, |
| "completions/min_length": 14.3, |
| "completions/min_terminated_length": 14.3, |
| "entropy": 0.05750619564205408, |
| "epoch": 0.2, |
| "frac_reward_zero_std": 0.1, |
| "grad_norm": 2.755150318145752, |
| "learning_rate": 1.655e-06, |
| "loss": -0.009651938825845719, |
| "num_tokens": 738518.0, |
| "reward": 0.7117187455296516, |
| "reward_std": 0.2861814171075821, |
| "rewards/<lambda>/mean": 0.7117187455296516, |
| "rewards/<lambda>/std": 0.2861814320087433, |
| "step": 70, |
| "step_time": 8.00788230150938 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.83125, |
| "completions/mean_terminated_length": 15.83125, |
| "completions/min_length": 14.7, |
| "completions/min_terminated_length": 14.7, |
| "entropy": 0.05246723517775535, |
| "epoch": 0.22857142857142856, |
| "frac_reward_zero_std": 0.6, |
| "grad_norm": 2.081681728363037, |
| "learning_rate": 1.6049999999999999e-06, |
| "loss": 0.0006721832789480687, |
| "num_tokens": 832891.0, |
| "reward": 0.7782250106334686, |
| "reward_std": 0.06980862021446228, |
| "rewards/<lambda>/mean": 0.7782250106334686, |
| "rewards/<lambda>/std": 0.06980862021446228, |
| "step": 80, |
| "step_time": 7.233118758106139 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.0, |
| "completions/max_terminated_length": 16.0, |
| "completions/mean_length": 15.94375, |
| "completions/mean_terminated_length": 15.94375, |
| "completions/min_length": 15.2, |
| "completions/min_terminated_length": 15.2, |
| "entropy": 0.03160495152696967, |
| "epoch": 0.2571428571428571, |
| "frac_reward_zero_std": 0.7, |
| "grad_norm": 3.393371343612671, |
| "learning_rate": 1.555e-06, |
| "loss": 7.450580596923828e-09, |
| "num_tokens": 948722.0, |
| "reward": 0.8057500004768372, |
| "reward_std": 0.07212598472833634, |
| "rewards/<lambda>/mean": 0.8057500004768372, |
| "rewards/<lambda>/std": 0.07212599217891694, |
| "step": 90, |
| "step_time": 8.612589649111033 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0, |
| "completions/max_length": 16.5, |
| "completions/max_terminated_length": 16.5, |
| "completions/mean_length": 16.025, |
| "completions/mean_terminated_length": 16.025, |
| "completions/min_length": 15.9, |
| "completions/min_terminated_length": 15.9, |
| "entropy": 0.01998692280612886, |
| "epoch": 0.2857142857142857, |
| "frac_reward_zero_std": 0.3, |
| "grad_norm": 3.434286117553711, |
| "learning_rate": 1.5049999999999998e-06, |
| "loss": 0.007304742932319641, |
| "num_tokens": 1066742.0, |
| "reward": 0.9358687460422516, |
| "reward_std": 0.14476753100752832, |
| "rewards/<lambda>/mean": 0.9358687460422516, |
| "rewards/<lambda>/std": 0.14476753398776054, |
| "step": 100, |
| "step_time": 8.702565377613064 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 400, |
| "num_input_tokens_seen": 1066742, |
| "num_train_epochs": 2, |
| "save_steps": 25, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 16, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|