stefanj0's picture
Training in progress, step 800, checkpoint
cf0eea4 verified
Raw
History Blame Contribute Delete
4.68 kB
{
"best_global_step": 200,
"best_metric": 0.6018396615982056,
"best_model_checkpoint": "./t5gemma-math-corrector/checkpoint-200",
"epoch": 2.5645549318364074,
"eval_steps": 200,
"global_step": 800,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.16038492381716118,
"grad_norm": 2.080294370651245,
"learning_rate": 5.212765957446809e-05,
"loss": 1.8373,
"step": 50
},
{
"epoch": 0.32076984763432237,
"grad_norm": 1.3640859127044678,
"learning_rate": 9.94061757719715e-05,
"loss": 0.028,
"step": 100
},
{
"epoch": 0.48115477145148355,
"grad_norm": 0.47297751903533936,
"learning_rate": 9.346793349168646e-05,
"loss": 0.0138,
"step": 150
},
{
"epoch": 0.6415396952686447,
"grad_norm": 0.6050183176994324,
"learning_rate": 8.752969121140144e-05,
"loss": 0.0085,
"step": 200
},
{
"epoch": 0.6415396952686447,
"eval_loss": 0.6018396615982056,
"eval_runtime": 5.1929,
"eval_samples_per_second": 50.454,
"eval_steps_per_second": 12.71,
"step": 200
},
{
"epoch": 0.8019246190858059,
"grad_norm": 0.2022608071565628,
"learning_rate": 8.15914489311164e-05,
"loss": 0.0064,
"step": 250
},
{
"epoch": 0.9623095429029671,
"grad_norm": 0.8748292922973633,
"learning_rate": 7.565320665083135e-05,
"loss": 0.0047,
"step": 300
},
{
"epoch": 1.1218925421010426,
"grad_norm": 0.28713762760162354,
"learning_rate": 6.971496437054633e-05,
"loss": 0.0025,
"step": 350
},
{
"epoch": 1.2822774659182037,
"grad_norm": 0.16400396823883057,
"learning_rate": 6.377672209026129e-05,
"loss": 0.0029,
"step": 400
},
{
"epoch": 1.2822774659182037,
"eval_loss": 1.9704385995864868,
"eval_runtime": 5.1665,
"eval_samples_per_second": 50.711,
"eval_steps_per_second": 12.775,
"step": 400
},
{
"epoch": 1.4426623897353648,
"grad_norm": 0.35608166456222534,
"learning_rate": 5.783847980997625e-05,
"loss": 0.0015,
"step": 450
},
{
"epoch": 1.603047313552526,
"grad_norm": 0.33180880546569824,
"learning_rate": 5.190023752969121e-05,
"loss": 0.0015,
"step": 500
},
{
"epoch": 1.7634322373696873,
"grad_norm": 0.16627037525177002,
"learning_rate": 4.596199524940617e-05,
"loss": 0.0009,
"step": 550
},
{
"epoch": 1.9238171611868484,
"grad_norm": 0.04953346028923988,
"learning_rate": 4.002375296912114e-05,
"loss": 0.001,
"step": 600
},
{
"epoch": 1.9238171611868484,
"eval_loss": 3.8351235389709473,
"eval_runtime": 5.1639,
"eval_samples_per_second": 50.737,
"eval_steps_per_second": 12.781,
"step": 600
},
{
"epoch": 2.0834001603849237,
"grad_norm": 0.13095279037952423,
"learning_rate": 3.408551068883611e-05,
"loss": 0.0009,
"step": 650
},
{
"epoch": 2.2437850842020852,
"grad_norm": 0.1417347490787506,
"learning_rate": 2.8147268408551068e-05,
"loss": 0.0005,
"step": 700
},
{
"epoch": 2.4041700080192463,
"grad_norm": 0.07039827853441238,
"learning_rate": 2.2209026128266035e-05,
"loss": 0.0005,
"step": 750
},
{
"epoch": 2.5645549318364074,
"grad_norm": 0.2078932374715805,
"learning_rate": 1.6270783847980998e-05,
"loss": 0.0005,
"step": 800
},
{
"epoch": 2.5645549318364074,
"eval_loss": 3.8249130249023438,
"eval_runtime": 5.2131,
"eval_samples_per_second": 50.258,
"eval_steps_per_second": 12.66,
"step": 800
}
],
"logging_steps": 50,
"max_steps": 936,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 200,
"stateful_callbacks": {
"EarlyStoppingCallback": {
"args": {
"early_stopping_patience": 3,
"early_stopping_threshold": 0.0
},
"attributes": {
"early_stopping_patience_counter": 3
}
},
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3090020422385664.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}