hs-zz27's picture
Upload folder using huggingface_hub
612514a verified
Raw
History Blame Contribute Delete
10.6 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.2857142857142857,
"eval_steps": 500,
"global_step": 100,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.1,
"completions/max_terminated_length": 16.1,
"completions/mean_length": 15.43125,
"completions/mean_terminated_length": 15.43125,
"completions/min_length": 12.8,
"completions/min_terminated_length": 12.8,
"entropy": 0.11648641154170036,
"epoch": 0.02857142857142857,
"frac_reward_zero_std": 0.0,
"grad_norm": 3.766094207763672,
"learning_rate": 1.955e-06,
"loss": -0.03027614951133728,
"num_tokens": 117861.0,
"reward": 0.42463125213980674,
"reward_std": 0.4041272960603237,
"rewards/<lambda>/mean": 0.42463125213980674,
"rewards/<lambda>/std": 0.4041273109614849,
"step": 10,
"step_time": 8.86154415559722
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 39.1,
"completions/max_terminated_length": 39.1,
"completions/mean_length": 15.76875,
"completions/mean_terminated_length": 15.76875,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"entropy": 0.14287759065628053,
"epoch": 0.05714285714285714,
"frac_reward_zero_std": 0.3,
"grad_norm": 4.652894020080566,
"learning_rate": 1.905e-06,
"loss": -0.007219273597002029,
"num_tokens": 211120.0,
"reward": 0.16723437011241912,
"reward_std": 0.23346474170684814,
"rewards/<lambda>/mean": 0.16723437011241912,
"rewards/<lambda>/std": 0.233464752137661,
"step": 20,
"step_time": 8.861151697207243
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.6,
"completions/max_terminated_length": 16.6,
"completions/mean_length": 15.18125,
"completions/mean_terminated_length": 15.18125,
"completions/min_length": 12.6,
"completions/min_terminated_length": 12.6,
"entropy": 0.10504549741744995,
"epoch": 0.08571428571428572,
"frac_reward_zero_std": 0.1,
"grad_norm": 2.999187707901001,
"learning_rate": 1.8549999999999998e-06,
"loss": -0.01587933599948883,
"num_tokens": 315101.0,
"reward": 0.5093562602996826,
"reward_std": 0.2611938640475273,
"rewards/<lambda>/mean": 0.5093562602996826,
"rewards/<lambda>/std": 0.26119386702775954,
"step": 30,
"step_time": 8.02011652730871
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.83125,
"completions/mean_terminated_length": 15.83125,
"completions/min_length": 14.6,
"completions/min_terminated_length": 14.6,
"entropy": 0.07781314179301262,
"epoch": 0.11428571428571428,
"frac_reward_zero_std": 0.0,
"grad_norm": 2.083005666732788,
"learning_rate": 1.8049999999999999e-06,
"loss": -0.0016902854666113853,
"num_tokens": 431602.0,
"reward": 0.7693868696689605,
"reward_std": 0.3072405755519867,
"rewards/<lambda>/mean": 0.7693868696689605,
"rewards/<lambda>/std": 0.3072405830025673,
"step": 40,
"step_time": 8.891476768592838
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.36875,
"completions/mean_terminated_length": 15.36875,
"completions/min_length": 13.9,
"completions/min_terminated_length": 13.9,
"entropy": 0.09770065676420928,
"epoch": 0.14285714285714285,
"frac_reward_zero_std": 0.2,
"grad_norm": 0.0,
"learning_rate": 1.7549999999999997e-06,
"loss": -0.004019740223884583,
"num_tokens": 525533.0,
"reward": 0.5823375027626753,
"reward_std": 0.2531014457345009,
"rewards/<lambda>/mean": 0.5823375027626753,
"rewards/<lambda>/std": 0.25310145914554594,
"step": 50,
"step_time": 7.289726911496837
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.8875,
"completions/mean_terminated_length": 15.8875,
"completions/min_length": 15.5,
"completions/min_terminated_length": 15.5,
"entropy": 0.058196247555315495,
"epoch": 0.17142857142857143,
"frac_reward_zero_std": 0.2,
"grad_norm": 3.2708792686462402,
"learning_rate": 1.705e-06,
"loss": 0.0007023245096206665,
"num_tokens": 633339.0,
"reward": 0.8079562425613404,
"reward_std": 0.26226545721292494,
"rewards/<lambda>/mean": 0.8079562425613404,
"rewards/<lambda>/std": 0.26226547062397004,
"step": 60,
"step_time": 8.16725606439868
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.86875,
"completions/mean_terminated_length": 15.86875,
"completions/min_length": 14.3,
"completions/min_terminated_length": 14.3,
"entropy": 0.05750619564205408,
"epoch": 0.2,
"frac_reward_zero_std": 0.1,
"grad_norm": 2.755150318145752,
"learning_rate": 1.655e-06,
"loss": -0.009651938825845719,
"num_tokens": 738518.0,
"reward": 0.7117187455296516,
"reward_std": 0.2861814171075821,
"rewards/<lambda>/mean": 0.7117187455296516,
"rewards/<lambda>/std": 0.2861814320087433,
"step": 70,
"step_time": 8.00788230150938
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.83125,
"completions/mean_terminated_length": 15.83125,
"completions/min_length": 14.7,
"completions/min_terminated_length": 14.7,
"entropy": 0.05246723517775535,
"epoch": 0.22857142857142856,
"frac_reward_zero_std": 0.6,
"grad_norm": 2.081681728363037,
"learning_rate": 1.6049999999999999e-06,
"loss": 0.0006721832789480687,
"num_tokens": 832891.0,
"reward": 0.7782250106334686,
"reward_std": 0.06980862021446228,
"rewards/<lambda>/mean": 0.7782250106334686,
"rewards/<lambda>/std": 0.06980862021446228,
"step": 80,
"step_time": 7.233118758106139
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.0,
"completions/max_terminated_length": 16.0,
"completions/mean_length": 15.94375,
"completions/mean_terminated_length": 15.94375,
"completions/min_length": 15.2,
"completions/min_terminated_length": 15.2,
"entropy": 0.03160495152696967,
"epoch": 0.2571428571428571,
"frac_reward_zero_std": 0.7,
"grad_norm": 3.393371343612671,
"learning_rate": 1.555e-06,
"loss": 7.450580596923828e-09,
"num_tokens": 948722.0,
"reward": 0.8057500004768372,
"reward_std": 0.07212598472833634,
"rewards/<lambda>/mean": 0.8057500004768372,
"rewards/<lambda>/std": 0.07212599217891694,
"step": 90,
"step_time": 8.612589649111033
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 16.5,
"completions/max_terminated_length": 16.5,
"completions/mean_length": 16.025,
"completions/mean_terminated_length": 16.025,
"completions/min_length": 15.9,
"completions/min_terminated_length": 15.9,
"entropy": 0.01998692280612886,
"epoch": 0.2857142857142857,
"frac_reward_zero_std": 0.3,
"grad_norm": 3.434286117553711,
"learning_rate": 1.5049999999999998e-06,
"loss": 0.007304742932319641,
"num_tokens": 1066742.0,
"reward": 0.9358687460422516,
"reward_std": 0.14476753100752832,
"rewards/<lambda>/mean": 0.9358687460422516,
"rewards/<lambda>/std": 0.14476753398776054,
"step": 100,
"step_time": 8.702565377613064
}
],
"logging_steps": 10,
"max_steps": 400,
"num_input_tokens_seen": 1066742,
"num_train_epochs": 2,
"save_steps": 25,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}