{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.018195050946142648, "eval_steps": 100, "global_step": 100, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0009097525473071324, "grad_norm": 1.0602493286132812, "learning_rate": 1.2121212121212122e-06, "loss": 1.7156932830810547, "step": 5 }, { "epoch": 0.001819505094614265, "grad_norm": 1.1577719449996948, "learning_rate": 2.7272727272727272e-06, "loss": 1.6629371643066406, "step": 10 }, { "epoch": 0.0027292576419213972, "grad_norm": 1.0288419723510742, "learning_rate": 4.242424242424243e-06, "loss": 1.6706295013427734, "step": 15 }, { "epoch": 0.00363901018922853, "grad_norm": 2.129403829574585, "learning_rate": 5.7575757575757586e-06, "loss": 1.7363752365112304, "step": 20 }, { "epoch": 0.004548762736535662, "grad_norm": 1.9468326568603516, "learning_rate": 7.272727272727272e-06, "loss": 1.7111135482788087, "step": 25 }, { "epoch": 0.0054585152838427945, "grad_norm": 1.1269357204437256, "learning_rate": 8.787878787878788e-06, "loss": 1.6924203872680663, "step": 30 }, { "epoch": 0.006368267831149927, "grad_norm": 1.4021248817443848, "learning_rate": 1.0303030303030304e-05, "loss": 1.658310317993164, "step": 35 }, { "epoch": 0.00727802037845706, "grad_norm": 1.313381314277649, "learning_rate": 1.1818181818181819e-05, "loss": 1.5383296012878418, "step": 40 }, { "epoch": 0.008187772925764192, "grad_norm": 2.4359891414642334, "learning_rate": 1.3333333333333333e-05, "loss": 1.4302565574645996, "step": 45 }, { "epoch": 0.009097525473071324, "grad_norm": 1.6459542512893677, "learning_rate": 1.484848484848485e-05, "loss": 1.2602953910827637, "step": 50 }, { "epoch": 0.010007278020378457, "grad_norm": 0.7953159213066101, "learning_rate": 1.6363636363636366e-05, "loss": 1.204326343536377, "step": 55 }, { "epoch": 0.010917030567685589, "grad_norm": 0.5824465155601501, "learning_rate": 1.787878787878788e-05, "loss": 1.068561840057373, "step": 60 }, { "epoch": 0.011826783114992722, "grad_norm": 0.39265626668930054, "learning_rate": 1.9393939393939395e-05, "loss": 0.9570062637329102, "step": 65 }, { "epoch": 0.012736535662299854, "grad_norm": 0.3387283384799957, "learning_rate": 2.090909090909091e-05, "loss": 0.9454713821411133, "step": 70 }, { "epoch": 0.013646288209606987, "grad_norm": 0.3182811141014099, "learning_rate": 2.2424242424242424e-05, "loss": 0.8901592254638672, "step": 75 }, { "epoch": 0.01455604075691412, "grad_norm": 0.2735312879085541, "learning_rate": 2.393939393939394e-05, "loss": 0.8491583824157715, "step": 80 }, { "epoch": 0.015465793304221253, "grad_norm": 0.2376435250043869, "learning_rate": 2.5454545454545454e-05, "loss": 0.8109179496765136, "step": 85 }, { "epoch": 0.016375545851528384, "grad_norm": 0.2161586880683899, "learning_rate": 2.696969696969697e-05, "loss": 0.76962308883667, "step": 90 }, { "epoch": 0.017285298398835518, "grad_norm": 0.19587980210781097, "learning_rate": 2.8484848484848486e-05, "loss": 0.7301986694335938, "step": 95 }, { "epoch": 0.018195050946142648, "grad_norm": 0.20971694588661194, "learning_rate": 3e-05, "loss": 0.7269618034362793, "step": 100 }, { "epoch": 0.018195050946142648, "eval_loss": 2.605874538421631, "eval_runtime": 1120.0905, "eval_samples_per_second": 33.935, "eval_steps_per_second": 8.484, "step": 100 } ], "logging_steps": 5, "max_steps": 5500, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 6.444622973392128e+16, "train_batch_size": 8, "trial_name": null, "trial_params": null }