{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 846, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.035555555555555556, "grad_norm": 0.7644901871681213, "learning_rate": 2.1176470588235296e-05, "loss": 0.30520846843719485, "step": 10 }, { "epoch": 0.07111111111111111, "grad_norm": 0.43560856580734253, "learning_rate": 4.470588235294118e-05, "loss": 0.26438264846801757, "step": 20 }, { "epoch": 0.10666666666666667, "grad_norm": 0.2563800811767578, "learning_rate": 6.823529411764707e-05, "loss": 0.19181081056594848, "step": 30 }, { "epoch": 0.14222222222222222, "grad_norm": 0.10259224474430084, "learning_rate": 9.176470588235295e-05, "loss": 0.15193926095962523, "step": 40 }, { "epoch": 0.17777777777777778, "grad_norm": 0.1993834525346756, "learning_rate": 0.00011529411764705881, "loss": 0.13089948892593384, "step": 50 }, { "epoch": 0.21333333333333335, "grad_norm": 0.19039888679981232, "learning_rate": 0.00013882352941176472, "loss": 0.1438336730003357, "step": 60 }, { "epoch": 0.24888888888888888, "grad_norm": 0.153082013130188, "learning_rate": 0.0001623529411764706, "loss": 0.13305349349975587, "step": 70 }, { "epoch": 0.28444444444444444, "grad_norm": 0.08913639187812805, "learning_rate": 0.00018588235294117648, "loss": 0.16499919891357423, "step": 80 }, { "epoch": 0.32, "grad_norm": 0.18134109675884247, "learning_rate": 0.00019998636639992777, "loss": 0.13644593954086304, "step": 90 }, { "epoch": 0.35555555555555557, "grad_norm": 0.09949612617492676, "learning_rate": 0.00019983303108908946, "loss": 0.13039498329162597, "step": 100 }, { "epoch": 0.39111111111111113, "grad_norm": 0.12579314410686493, "learning_rate": 0.00019950958062149127, "loss": 0.1297929048538208, "step": 110 }, { "epoch": 0.4266666666666667, "grad_norm": 0.1197328120470047, "learning_rate": 0.00019901656615566656, "loss": 0.1509210705757141, "step": 120 }, { "epoch": 0.4622222222222222, "grad_norm": 0.15709449350833893, "learning_rate": 0.00019835482778664425, "loss": 0.14251822233200073, "step": 130 }, { "epoch": 0.49777777777777776, "grad_norm": 0.2203192412853241, "learning_rate": 0.0001975254931144296, "loss": 0.1412465214729309, "step": 140 }, { "epoch": 0.5333333333333333, "grad_norm": 0.32334810495376587, "learning_rate": 0.0001965299753225775, "loss": 0.1519417643547058, "step": 150 }, { "epoch": 0.5688888888888889, "grad_norm": 0.1700880080461502, "learning_rate": 0.00019536997077013236, "loss": 0.13457289934158326, "step": 160 }, { "epoch": 0.6044444444444445, "grad_norm": 0.17782148718833923, "learning_rate": 0.00019404745610103786, "loss": 0.14420714378356933, "step": 170 }, { "epoch": 0.64, "grad_norm": 0.19534672796726227, "learning_rate": 0.00019256468487594214, "loss": 0.1206929087638855, "step": 180 }, { "epoch": 0.6755555555555556, "grad_norm": 0.12961238622665405, "learning_rate": 0.00019092418373213796, "loss": 0.13416168689727784, "step": 190 }, { "epoch": 0.7111111111111111, "grad_norm": 0.19645075500011444, "learning_rate": 0.000189128748078181, "loss": 0.13229730129241943, "step": 200 }, { "epoch": 0.7466666666666667, "grad_norm": 0.17324262857437134, "learning_rate": 0.00018718143733052278, "loss": 0.12280937433242797, "step": 210 }, { "epoch": 0.7822222222222223, "grad_norm": 0.10764078050851822, "learning_rate": 0.0001850855697002753, "loss": 0.12583138942718505, "step": 220 }, { "epoch": 0.8177777777777778, "grad_norm": 0.13804763555526733, "learning_rate": 0.00018284471653898994, "loss": 0.13474913835525512, "step": 230 }, { "epoch": 0.8533333333333334, "grad_norm": 0.11816427856683731, "learning_rate": 0.00018046269625308648, "loss": 0.13484352827072144, "step": 240 }, { "epoch": 0.8888888888888888, "grad_norm": 0.09677577018737793, "learning_rate": 0.00017794356779730084, "loss": 0.14849199056625367, "step": 250 }, { "epoch": 0.9244444444444444, "grad_norm": 0.10201010853052139, "learning_rate": 0.00017529162375823958, "loss": 0.12720253467559814, "step": 260 }, { "epoch": 0.96, "grad_norm": 0.23005172610282898, "learning_rate": 0.00017251138303982675, "loss": 0.11229866743087769, "step": 270 }, { "epoch": 0.9955555555555555, "grad_norm": 0.15584969520568848, "learning_rate": 0.00016960758316310597, "loss": 0.12440342903137207, "step": 280 }, { "epoch": 1.0284444444444445, "grad_norm": 0.08643736690282822, "learning_rate": 0.0001665851721935205, "loss": 0.0838462769985199, "step": 290 }, { "epoch": 1.064, "grad_norm": 0.061173878610134125, "learning_rate": 0.0001634493003094259, "loss": 0.05495935678482056, "step": 300 }, { "epoch": 1.0995555555555556, "grad_norm": 0.3276287019252777, "learning_rate": 0.00016020531102620304, "loss": 0.055119764804840085, "step": 310 }, { "epoch": 1.1351111111111112, "grad_norm": 0.1756429374217987, "learning_rate": 0.0001568587320909255, "loss": 0.0650447130203247, "step": 320 }, { "epoch": 1.1706666666666667, "grad_norm": 0.1355263888835907, "learning_rate": 0.00015341526606309645, "loss": 0.06544734239578247, "step": 330 }, { "epoch": 1.2062222222222223, "grad_norm": 0.20916008949279785, "learning_rate": 0.00014988078059750652, "loss": 0.06727538704872131, "step": 340 }, { "epoch": 1.2417777777777779, "grad_norm": 0.19566652178764343, "learning_rate": 0.00014626129844576893, "loss": 0.06408223509788513, "step": 350 }, { "epoch": 1.2773333333333334, "grad_norm": 0.05987081304192543, "learning_rate": 0.00014256298719357062, "loss": 0.04535002112388611, "step": 360 }, { "epoch": 1.3128888888888888, "grad_norm": 0.1902439296245575, "learning_rate": 0.00013879214875112665, "loss": 0.04922315180301666, "step": 370 }, { "epoch": 1.3484444444444446, "grad_norm": 0.25207415223121643, "learning_rate": 0.00013495520861474565, "loss": 0.060137057304382326, "step": 380 }, { "epoch": 1.384, "grad_norm": 0.34174537658691406, "learning_rate": 0.00013105870491780558, "loss": 0.05463656783103943, "step": 390 }, { "epoch": 1.4195555555555557, "grad_norm": 0.19624291360378265, "learning_rate": 0.00012710927728979568, "loss": 0.06046912670135498, "step": 400 }, { "epoch": 1.455111111111111, "grad_norm": 0.22849531471729279, "learning_rate": 0.00012311365554240971, "loss": 0.039173880219459535, "step": 410 }, { "epoch": 1.4906666666666666, "grad_norm": 0.22941145300865173, "learning_rate": 0.0001190786482019691, "loss": 0.04884783029556274, "step": 420 }, { "epoch": 1.5262222222222221, "grad_norm": 0.2867370843887329, "learning_rate": 0.00011501113090771619, "loss": 0.05086652636528015, "step": 430 }, { "epoch": 1.561777777777778, "grad_norm": 0.27643826603889465, "learning_rate": 0.00011091803469574789, "loss": 0.04540249407291412, "step": 440 }, { "epoch": 1.5973333333333333, "grad_norm": 0.1365540772676468, "learning_rate": 0.00010680633418855267, "loss": 0.0644813358783722, "step": 450 }, { "epoch": 1.6328888888888888, "grad_norm": 0.21614767611026764, "learning_rate": 0.00010268303571027696, "loss": 0.07455622553825378, "step": 460 }, { "epoch": 1.6684444444444444, "grad_norm": 0.08498027920722961, "learning_rate": 9.855516534797187e-05, "loss": 0.04775569438934326, "step": 470 }, { "epoch": 1.704, "grad_norm": 0.2029683142900467, "learning_rate": 9.442975697916372e-05, "loss": 0.04402676820755005, "step": 480 }, { "epoch": 1.7395555555555555, "grad_norm": 0.34462177753448486, "learning_rate": 9.031384028615004e-05, "loss": 0.05336908102035522, "step": 490 }, { "epoch": 1.775111111111111, "grad_norm": 0.05993572995066643, "learning_rate": 8.621442877744409e-05, "loss": 0.04580667018890381, "step": 500 }, { "epoch": 1.775111111111111, "eval_accuracy": 0.9426666666666668, "eval_loss": 0.15560266375541687, "eval_mcq_accuracy": 0.7222222222222222, "eval_runtime": 26.3716, "eval_samples_per_second": 17.064, "eval_steps_per_second": 4.285, "step": 500 }, { "epoch": 1.8106666666666666, "grad_norm": 0.3891923427581787, "learning_rate": 8.213850783677925e-05, "loss": 0.03394646048545837, "step": 510 }, { "epoch": 1.8462222222222222, "grad_norm": 0.1975124329328537, "learning_rate": 7.809302282003823e-05, "loss": 0.054677408933639524, "step": 520 }, { "epoch": 1.8817777777777778, "grad_norm": 0.1962609589099884, "learning_rate": 7.408486722038943e-05, "loss": 0.06665679812431335, "step": 530 }, { "epoch": 1.9173333333333333, "grad_norm": 0.2969353199005127, "learning_rate": 7.012087092179724e-05, "loss": 0.048915204405784604, "step": 540 }, { "epoch": 1.952888888888889, "grad_norm": 0.23223088681697845, "learning_rate": 6.620778856092227e-05, "loss": 0.04671376347541809, "step": 550 }, { "epoch": 1.9884444444444445, "grad_norm": 0.4542539417743683, "learning_rate": 6.235228801724253e-05, "loss": 0.05778223276138306, "step": 560 }, { "epoch": 2.021333333333333, "grad_norm": 0.09194125235080719, "learning_rate": 5.856093905100899e-05, "loss": 0.022845838963985444, "step": 570 }, { "epoch": 2.056888888888889, "grad_norm": 0.10670984536409378, "learning_rate": 5.4840202108395466e-05, "loss": 0.00663934126496315, "step": 580 }, { "epoch": 2.0924444444444443, "grad_norm": 0.015582868829369545, "learning_rate": 5.119641731291971e-05, "loss": 0.007281148433685302, "step": 590 }, { "epoch": 2.128, "grad_norm": 0.17645655572414398, "learning_rate": 4.7635793661893666e-05, "loss": 0.00572865828871727, "step": 600 }, { "epoch": 2.1635555555555555, "grad_norm": 0.46430906653404236, "learning_rate": 4.416439844631271e-05, "loss": 0.008638855814933778, "step": 610 }, { "epoch": 2.1991111111111112, "grad_norm": 0.011537984013557434, "learning_rate": 4.078814691221139e-05, "loss": 0.004736468568444252, "step": 620 }, { "epoch": 2.2346666666666666, "grad_norm": 0.00896433461457491, "learning_rate": 3.751279218110387e-05, "loss": 0.007740923762321472, "step": 630 }, { "epoch": 2.2702222222222224, "grad_norm": 0.002848990960046649, "learning_rate": 3.434391544668383e-05, "loss": 0.007571302354335785, "step": 640 }, { "epoch": 2.3057777777777777, "grad_norm": 0.01570839062333107, "learning_rate": 3.1286916464488505e-05, "loss": 0.006724107265472412, "step": 650 }, { "epoch": 2.3413333333333335, "grad_norm": 0.003965158946812153, "learning_rate": 2.8347004350733185e-05, "loss": 0.008114227652549743, "step": 660 }, { "epoch": 2.376888888888889, "grad_norm": 0.10036994516849518, "learning_rate": 2.55291887059944e-05, "loss": 0.0027894463390111925, "step": 670 }, { "epoch": 2.4124444444444446, "grad_norm": 0.1702636480331421, "learning_rate": 2.2838271078866714e-05, "loss": 0.0071901664137840274, "step": 680 }, { "epoch": 2.448, "grad_norm": 0.003337263595312834, "learning_rate": 2.0278836784140044e-05, "loss": 0.0033892091363668443, "step": 690 }, { "epoch": 2.4835555555555557, "grad_norm": 0.0054590729996562, "learning_rate": 1.785524708943802e-05, "loss": 0.002477823756635189, "step": 700 }, { "epoch": 2.519111111111111, "grad_norm": 0.02814805507659912, "learning_rate": 1.557163178363251e-05, "loss": 0.0032980531454086305, "step": 710 }, { "epoch": 2.554666666666667, "grad_norm": 0.01721755601465702, "learning_rate": 1.3431882139696916e-05, "loss": 0.00092921182513237, "step": 720 }, { "epoch": 2.590222222222222, "grad_norm": 0.004700134973973036, "learning_rate": 1.1439644283989747e-05, "loss": 0.008682060241699218, "step": 730 }, { "epoch": 2.6257777777777775, "grad_norm": 0.02318074181675911, "learning_rate": 9.59831298326731e-06, "loss": 0.0035014577209949494, "step": 740 }, { "epoch": 2.6613333333333333, "grad_norm": 0.10873377323150635, "learning_rate": 7.911025860012444e-06, "loss": 0.007874777913093567, "step": 750 }, { "epoch": 2.696888888888889, "grad_norm": 0.007307600695639849, "learning_rate": 6.38065804593595e-06, "loss": 0.003820793330669403, "step": 760 }, { "epoch": 2.7324444444444445, "grad_norm": 0.002649878617376089, "learning_rate": 5.009817282761675e-06, "loss": 0.0014227939769625663, "step": 770 }, { "epoch": 2.768, "grad_norm": 0.02097943052649498, "learning_rate": 3.800839478643259e-06, "loss": 0.007305952161550522, "step": 780 }, { "epoch": 2.8035555555555556, "grad_norm": 0.11966679990291595, "learning_rate": 2.7557847277841942e-06, "loss": 0.004338302463293075, "step": 790 }, { "epoch": 2.8391111111111114, "grad_norm": 0.07167425006628036, "learning_rate": 1.8764338000442083e-06, "loss": 0.002496876008808613, "step": 800 }, { "epoch": 2.8746666666666667, "grad_norm": 0.010830031707882881, "learning_rate": 1.1642851065131633e-06, "loss": 0.0015433511696755886, "step": 810 }, { "epoch": 2.910222222222222, "grad_norm": 0.195694237947464, "learning_rate": 6.205521462235186e-07, "loss": 0.001970846764743328, "step": 820 }, { "epoch": 2.945777777777778, "grad_norm": 0.00489334249868989, "learning_rate": 2.4616143835202166e-07, "loss": 0.010426017642021179, "step": 830 }, { "epoch": 2.981333333333333, "grad_norm": 0.009853039868175983, "learning_rate": 4.1750943434026855e-08, "loss": 0.0070929393172264096, "step": 840 }, { "epoch": 3.0, "step": 846, "total_flos": 1.0958550940031386e+17, "train_loss": 0.06952556652285546, "train_runtime": 2162.4382, "train_samples_per_second": 6.243, "train_steps_per_second": 0.391 } ], "logging_steps": 10, "max_steps": 846, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 1.0958550940031386e+17, "train_batch_size": 4, "trial_name": null, "trial_params": null }