{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 2.0, "eval_steps": 500, "global_step": 886, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0451913571529445, "grad_norm": 1.2982341051101685, "learning_rate": 9.5e-05, "loss": 2.2395, "step": 20 }, { "epoch": 0.090382714305889, "grad_norm": 0.2602206766605377, "learning_rate": 0.000195, "loss": 0.3192, "step": 40 }, { "epoch": 0.1355740714588335, "grad_norm": 0.11576834321022034, "learning_rate": 0.000199892800315311, "loss": 0.0387, "step": 60 }, { "epoch": 0.180765428611778, "grad_norm": 0.08919216692447662, "learning_rate": 0.0001995485952541953, "loss": 0.0283, "step": 80 }, { "epoch": 0.2259567857647225, "grad_norm": 0.07938820868730545, "learning_rate": 0.00019896790549505508, "loss": 0.027, "step": 100 }, { "epoch": 0.271148142917667, "grad_norm": 0.10239128768444061, "learning_rate": 0.00019815211050730418, "loss": 0.0268, "step": 120 }, { "epoch": 0.3163395000706115, "grad_norm": 0.07505609840154648, "learning_rate": 0.00019710314826938252, "loss": 0.0265, "step": 140 }, { "epoch": 0.361530857223556, "grad_norm": 0.07303870469331741, "learning_rate": 0.00019582351066495192, "loss": 0.0263, "step": 160 }, { "epoch": 0.40672221437650047, "grad_norm": 0.0660228356719017, "learning_rate": 0.0001943162375632511, "loss": 0.0262, "step": 180 }, { "epoch": 0.451913571529445, "grad_norm": 0.06548438221216202, "learning_rate": 0.000192584909597672, "loss": 0.026, "step": 200 }, { "epoch": 0.4971049286823895, "grad_norm": 0.06231554225087166, "learning_rate": 0.0001906336396597133, "loss": 0.0257, "step": 220 }, { "epoch": 0.542296285835334, "grad_norm": 0.05514415726065636, "learning_rate": 0.00018846706312851684, "loss": 0.026, "step": 240 }, { "epoch": 0.5874876429882785, "grad_norm": 0.04507150128483772, "learning_rate": 0.00018609032685919813, "loss": 0.026, "step": 260 }, { "epoch": 0.632679000141223, "grad_norm": 0.04801488667726517, "learning_rate": 0.00018350907695612961, "loss": 0.0257, "step": 280 }, { "epoch": 0.6778703572941674, "grad_norm": 0.05931699648499489, "learning_rate": 0.00018072944536022211, "loss": 0.0254, "step": 300 }, { "epoch": 0.723061714447112, "grad_norm": 0.04856206849217415, "learning_rate": 0.00017775803528206736, "loss": 0.0255, "step": 320 }, { "epoch": 0.7682530716000565, "grad_norm": 0.04942520335316658, "learning_rate": 0.00017460190551554634, "loss": 0.0256, "step": 340 }, { "epoch": 0.8134444287530009, "grad_norm": 0.04102044925093651, "learning_rate": 0.0001712685536691673, "loss": 0.0254, "step": 360 }, { "epoch": 0.8586357859059455, "grad_norm": 0.04938124492764473, "learning_rate": 0.00016776589835496876, "loss": 0.0258, "step": 380 }, { "epoch": 0.90382714305889, "grad_norm": 0.03212631866335869, "learning_rate": 0.00016410226037729908, "loss": 0.0256, "step": 400 }, { "epoch": 0.9490185002118345, "grad_norm": 0.03286610171198845, "learning_rate": 0.00016028634296615972, "loss": 0.0254, "step": 420 }, { "epoch": 0.994209857364779, "grad_norm": 0.03531966730952263, "learning_rate": 0.00015632721110206958, "loss": 0.0252, "step": 440 }, { "epoch": 1.038412653580003, "grad_norm": 0.046482112258672714, "learning_rate": 0.0001522342699815653, "loss": 0.0257, "step": 460 }, { "epoch": 1.0836040107329472, "grad_norm": 0.05457223206758499, "learning_rate": 0.00014801724267449477, "loss": 0.0256, "step": 480 }, { "epoch": 1.1287953678858917, "grad_norm": 0.040691521018743515, "learning_rate": 0.00014368614702617996, "loss": 0.0248, "step": 500 }, { "epoch": 1.1739867250388363, "grad_norm": 0.042198147624731064, "learning_rate": 0.00013925127185931993, "loss": 0.0254, "step": 520 }, { "epoch": 1.2191780821917808, "grad_norm": 0.03628581762313843, "learning_rate": 0.0001347231525321678, "loss": 0.0253, "step": 540 }, { "epoch": 1.2643694393447253, "grad_norm": 0.038468316197395325, "learning_rate": 0.00013011254591104577, "loss": 0.0252, "step": 560 }, { "epoch": 1.3095607964976699, "grad_norm": 0.046080492436885834, "learning_rate": 0.00012543040481665133, "loss": 0.0256, "step": 580 }, { "epoch": 1.3547521536506144, "grad_norm": 0.03800000995397568, "learning_rate": 0.00012068785200486044, "loss": 0.0254, "step": 600 }, { "epoch": 1.399943510803559, "grad_norm": 0.03493132442235947, "learning_rate": 0.00011589615374383794, "loss": 0.0253, "step": 620 }, { "epoch": 1.4451348679565033, "grad_norm": 0.04193327575922012, "learning_rate": 0.00011106669305022396, "loss": 0.025, "step": 640 }, { "epoch": 1.4903262251094478, "grad_norm": 0.03124556504189968, "learning_rate": 0.00010621094264797647, "loss": 0.0254, "step": 660 }, { "epoch": 1.5355175822623923, "grad_norm": 0.03751407563686371, "learning_rate": 0.00010134043771410743, "loss": 0.0252, "step": 680 }, { "epoch": 1.5807089394153369, "grad_norm": 0.034240107983350754, "learning_rate": 9.646674847605779e-05, "loss": 0.0251, "step": 700 }, { "epoch": 1.6259002965682812, "grad_norm": 0.03505250811576843, "learning_rate": 9.160145272580727e-05, "loss": 0.0249, "step": 720 }, { "epoch": 1.6710916537212257, "grad_norm": 0.03779418393969536, "learning_rate": 8.675610831601423e-05, "loss": 0.025, "step": 740 }, { "epoch": 1.7162830108741702, "grad_norm": 0.037357985973358154, "learning_rate": 8.194222570352295e-05, "loss": 0.025, "step": 760 }, { "epoch": 1.7614743680271148, "grad_norm": 0.03674224391579628, "learning_rate": 7.717124060546254e-05, "loss": 0.0249, "step": 780 }, { "epoch": 1.8066657251800593, "grad_norm": 0.03422726318240166, "learning_rate": 7.245448683289604e-05, "loss": 0.0248, "step": 800 }, { "epoch": 1.8518570823330038, "grad_norm": 0.03446003794670105, "learning_rate": 6.780316936655382e-05, "loss": 0.0249, "step": 820 }, { "epoch": 1.8970484394859484, "grad_norm": 0.03549942001700401, "learning_rate": 6.322833773861297e-05, "loss": 0.0248, "step": 840 }, { "epoch": 1.942239796638893, "grad_norm": 0.03940591961145401, "learning_rate": 5.874085978375547e-05, "loss": 0.0249, "step": 860 }, { "epoch": 1.9874311537918374, "grad_norm": 0.029423682019114494, "learning_rate": 5.4351395821861665e-05, "loss": 0.0248, "step": 880 } ], "logging_steps": 20, "max_steps": 1329, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 6.363379718268518e+17, "train_batch_size": 2, "trial_name": null, "trial_params": null }