Invalid JSON:Unexpected token 'I', ..."ad_norm": Infinity,
"... is not valid JSON
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.36363636363636365, | |
| "eval_steps": 100, | |
| "global_step": 4000, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0009090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.999983482695944e-07, | |
| "loss": 2.972582244873047, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.0018181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.999926385982519e-07, | |
| "loss": 3.030682182312012, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.0027272727272727275, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.999828506408003e-07, | |
| "loss": 2.846811294555664, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.0036363636363636364, | |
| "grad_norm": 1.1991735607408722e+19, | |
| "learning_rate": 9.999689844770766e-07, | |
| "loss": 2.7316921234130858, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.004545454545454545, | |
| "grad_norm": 6.335976986745242e+18, | |
| "learning_rate": 9.999510402201835e-07, | |
| "loss": 2.9967422485351562, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.005454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.99929018016486e-07, | |
| "loss": 2.680373191833496, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.006363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.999029180456129e-07, | |
| "loss": 2.49072265625, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.007272727272727273, | |
| "grad_norm": 1.1419123145802514e+19, | |
| "learning_rate": 9.998727405204534e-07, | |
| "loss": 2.3953088760375976, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.008181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.998384856871564e-07, | |
| "loss": 2.656139945983887, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.00909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.998001538251282e-07, | |
| "loss": 2.8163694381713866, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.00909090909090909, | |
| "eval_loss": 2.7871742248535156, | |
| "eval_runtime": 25.2604, | |
| "eval_samples_per_second": 5.819, | |
| "eval_steps_per_second": 1.465, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.997577452470298e-07, | |
| "loss": 2.816670608520508, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.01090909090909091, | |
| "grad_norm": 1.794573070979249e+19, | |
| "learning_rate": 9.99711260298775e-07, | |
| "loss": 2.6952537536621093, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.011818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.99660699359527e-07, | |
| "loss": 2.715743064880371, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.012727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.996060628416963e-07, | |
| "loss": 3.068762016296387, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.013636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.995473511909357e-07, | |
| "loss": 2.851385498046875, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.014545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.994845648861382e-07, | |
| "loss": 2.8277969360351562, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.015454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.994177044394322e-07, | |
| "loss": 2.8090538024902343, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.016363636363636365, | |
| "grad_norm": 5.987306456414683e+18, | |
| "learning_rate": 9.993467703961784e-07, | |
| "loss": 2.7782918930053713, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.017272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.992717633349635e-07, | |
| "loss": 2.7075550079345705, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.01818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.99192683867597e-07, | |
| "loss": 3.050520133972168, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.01818181818181818, | |
| "eval_loss": 2.787334680557251, | |
| "eval_runtime": 25.3811, | |
| "eval_samples_per_second": 5.792, | |
| "eval_steps_per_second": 1.458, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.019090909090909092, | |
| "grad_norm": 2.708298899595985e+18, | |
| "learning_rate": 9.99109532639106e-07, | |
| "loss": 2.5547176361083985, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.99022310327729e-07, | |
| "loss": 2.71762638092041, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.02090909090909091, | |
| "grad_norm": 1.4361367883219468e+19, | |
| "learning_rate": 9.98931017644912e-07, | |
| "loss": 2.555472564697266, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.02181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.988356553353017e-07, | |
| "loss": 3.093941497802734, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.022727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.987362241767383e-07, | |
| "loss": 3.011051368713379, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.023636363636363636, | |
| "grad_norm": 5.583390414590706e+18, | |
| "learning_rate": 9.986327249802515e-07, | |
| "loss": 2.736302947998047, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.024545454545454544, | |
| "grad_norm": 1.3582739927916872e+19, | |
| "learning_rate": 9.985251585900525e-07, | |
| "loss": 2.630130577087402, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.025454545454545455, | |
| "grad_norm": 4.937351466969989e+18, | |
| "learning_rate": 9.984135258835272e-07, | |
| "loss": 2.6608613967895507, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.026363636363636363, | |
| "grad_norm": 1.5744615083612832e+19, | |
| "learning_rate": 9.98297827771229e-07, | |
| "loss": 2.79409236907959, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.02727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.981780651968722e-07, | |
| "loss": 2.7820377349853516, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.02727272727272727, | |
| "eval_loss": 2.7874202728271484, | |
| "eval_runtime": 25.2656, | |
| "eval_samples_per_second": 5.818, | |
| "eval_steps_per_second": 1.464, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.028181818181818183, | |
| "grad_norm": 5.150588553037611e+18, | |
| "learning_rate": 9.980542391373232e-07, | |
| "loss": 2.862852668762207, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.02909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.979263506025929e-07, | |
| "loss": 2.890887451171875, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.977944006358285e-07, | |
| "loss": 2.572690963745117, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.03090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.976583903133059e-07, | |
| "loss": 2.8408248901367186, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.031818181818181815, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.975183207444189e-07, | |
| "loss": 2.8312393188476563, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.03272727272727273, | |
| "grad_norm": 5.449467699304333e+18, | |
| "learning_rate": 9.97374193071672e-07, | |
| "loss": 3.0894535064697264, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.03363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.972260084706704e-07, | |
| "loss": 2.8833297729492187, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.034545454545454546, | |
| "grad_norm": 5.602140936134984e+18, | |
| "learning_rate": 9.970737681501104e-07, | |
| "loss": 3.0639154434204103, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.035454545454545454, | |
| "grad_norm": 6.436255195977482e+18, | |
| "learning_rate": 9.969174733517693e-07, | |
| "loss": 2.7005983352661134, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.03636363636363636, | |
| "grad_norm": 6.883443067668398e+18, | |
| "learning_rate": 9.967571253504956e-07, | |
| "loss": 2.7406898498535157, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.03636363636363636, | |
| "eval_loss": 2.787506580352783, | |
| "eval_runtime": 25.3731, | |
| "eval_samples_per_second": 5.794, | |
| "eval_steps_per_second": 1.458, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.03727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.965927254541986e-07, | |
| "loss": 3.2558399200439454, | |
| "step": 410 | |
| }, | |
| { | |
| "epoch": 0.038181818181818185, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.964242750038379e-07, | |
| "loss": 3.5612926483154297, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.03909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.96251775373412e-07, | |
| "loss": 2.661521339416504, | |
| "step": 430 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.960752279699469e-07, | |
| "loss": 2.7075674057006838, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 0.04090909090909091, | |
| "grad_norm": 1.3481015311138292e+19, | |
| "learning_rate": 9.958946342334858e-07, | |
| "loss": 2.752682685852051, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.04181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.957099956370761e-07, | |
| "loss": 3.051430892944336, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 0.042727272727272725, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.95521313686758e-07, | |
| "loss": 2.702523040771484, | |
| "step": 470 | |
| }, | |
| { | |
| "epoch": 0.04363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.953285899215524e-07, | |
| "loss": 2.804890251159668, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 0.04454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.951318259134473e-07, | |
| "loss": 2.9871782302856444, | |
| "step": 490 | |
| }, | |
| { | |
| "epoch": 0.045454545454545456, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.949310232673867e-07, | |
| "loss": 2.5105640411376955, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.045454545454545456, | |
| "eval_loss": 2.7876086235046387, | |
| "eval_runtime": 25.1494, | |
| "eval_samples_per_second": 5.845, | |
| "eval_steps_per_second": 1.471, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.046363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.947261836212555e-07, | |
| "loss": 2.7869237899780273, | |
| "step": 510 | |
| }, | |
| { | |
| "epoch": 0.04727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.94517308645868e-07, | |
| "loss": 2.9804574966430666, | |
| "step": 520 | |
| }, | |
| { | |
| "epoch": 0.04818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.94304400044953e-07, | |
| "loss": 2.777148628234863, | |
| "step": 530 | |
| }, | |
| { | |
| "epoch": 0.04909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.940874595551404e-07, | |
| "loss": 2.6402297973632813, | |
| "step": 540 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 6.372751802403652e+18, | |
| "learning_rate": 9.938664889459468e-07, | |
| "loss": 2.805499267578125, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 0.05090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.936414900197618e-07, | |
| "loss": 2.7490556716918944, | |
| "step": 560 | |
| }, | |
| { | |
| "epoch": 0.05181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.934124646118324e-07, | |
| "loss": 2.7342456817626952, | |
| "step": 570 | |
| }, | |
| { | |
| "epoch": 0.05272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.931794145902485e-07, | |
| "loss": 3.1899335861206053, | |
| "step": 580 | |
| }, | |
| { | |
| "epoch": 0.053636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.92942341855927e-07, | |
| "loss": 3.0976799011230467, | |
| "step": 590 | |
| }, | |
| { | |
| "epoch": 0.05454545454545454, | |
| "grad_norm": 5.517231700435796e+18, | |
| "learning_rate": 9.927012483425976e-07, | |
| "loss": 2.8919891357421874, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.05454545454545454, | |
| "eval_loss": 2.7876956462860107, | |
| "eval_runtime": 25.3974, | |
| "eval_samples_per_second": 5.788, | |
| "eval_steps_per_second": 1.457, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.05545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.924561360167858e-07, | |
| "loss": 2.845229721069336, | |
| "step": 610 | |
| }, | |
| { | |
| "epoch": 0.056363636363636366, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.922070068777972e-07, | |
| "loss": 2.4370473861694335, | |
| "step": 620 | |
| }, | |
| { | |
| "epoch": 0.057272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.91953862957702e-07, | |
| "loss": 2.5011491775512695, | |
| "step": 630 | |
| }, | |
| { | |
| "epoch": 0.05818181818181818, | |
| "grad_norm": 1.8244524918785966e+18, | |
| "learning_rate": 9.91696706321317e-07, | |
| "loss": 3.2636795043945312, | |
| "step": 640 | |
| }, | |
| { | |
| "epoch": 0.05909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.914355390661895e-07, | |
| "loss": 2.698811340332031, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 5.180378721080574e+18, | |
| "learning_rate": 9.91170363322581e-07, | |
| "loss": 2.5908447265625, | |
| "step": 660 | |
| }, | |
| { | |
| "epoch": 0.060909090909090906, | |
| "grad_norm": 2.5083164262012027e+18, | |
| "learning_rate": 9.90901181253448e-07, | |
| "loss": 2.9120254516601562, | |
| "step": 670 | |
| }, | |
| { | |
| "epoch": 0.06181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.906279950544258e-07, | |
| "loss": 2.828919219970703, | |
| "step": 680 | |
| }, | |
| { | |
| "epoch": 0.06272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.903508069538107e-07, | |
| "loss": 3.0443912506103517, | |
| "step": 690 | |
| }, | |
| { | |
| "epoch": 0.06363636363636363, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.900696192125401e-07, | |
| "loss": 2.940337562561035, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.06363636363636363, | |
| "eval_loss": 2.787841796875, | |
| "eval_runtime": 25.3047, | |
| "eval_samples_per_second": 5.809, | |
| "eval_steps_per_second": 1.462, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.06454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.897844341241764e-07, | |
| "loss": 2.8803565979003904, | |
| "step": 710 | |
| }, | |
| { | |
| "epoch": 0.06545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.894952540148861e-07, | |
| "loss": 2.391953468322754, | |
| "step": 720 | |
| }, | |
| { | |
| "epoch": 0.06636363636363636, | |
| "grad_norm": 7.620495189789901e+18, | |
| "learning_rate": 9.892020812434227e-07, | |
| "loss": 2.8717506408691404, | |
| "step": 730 | |
| }, | |
| { | |
| "epoch": 0.06727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.889049182011064e-07, | |
| "loss": 2.985699462890625, | |
| "step": 740 | |
| }, | |
| { | |
| "epoch": 0.06818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.886037673118046e-07, | |
| "loss": 2.728943634033203, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 0.06909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.882986310319123e-07, | |
| "loss": 2.5520145416259767, | |
| "step": 760 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 3.4245235227083407e+18, | |
| "learning_rate": 9.879895118503322e-07, | |
| "loss": 2.822847366333008, | |
| "step": 770 | |
| }, | |
| { | |
| "epoch": 0.07090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.876764122884546e-07, | |
| "loss": 2.851523780822754, | |
| "step": 780 | |
| }, | |
| { | |
| "epoch": 0.07181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.873593349001363e-07, | |
| "loss": 2.833907699584961, | |
| "step": 790 | |
| }, | |
| { | |
| "epoch": 0.07272727272727272, | |
| "grad_norm": 5.103755405008306e+18, | |
| "learning_rate": 9.870382822716797e-07, | |
| "loss": 2.9371898651123045, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.07272727272727272, | |
| "eval_loss": 2.7879319190979004, | |
| "eval_runtime": 25.258, | |
| "eval_samples_per_second": 5.82, | |
| "eval_steps_per_second": 1.465, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.07363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.867132570218127e-07, | |
| "loss": 2.694431114196777, | |
| "step": 810 | |
| }, | |
| { | |
| "epoch": 0.07454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.863842618016658e-07, | |
| "loss": 3.0608633041381834, | |
| "step": 820 | |
| }, | |
| { | |
| "epoch": 0.07545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.86051299294752e-07, | |
| "loss": 2.6543277740478515, | |
| "step": 830 | |
| }, | |
| { | |
| "epoch": 0.07636363636363637, | |
| "grad_norm": 1.7660894526035722e+19, | |
| "learning_rate": 9.857143722169442e-07, | |
| "loss": 2.9908905029296875, | |
| "step": 840 | |
| }, | |
| { | |
| "epoch": 0.07727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.853734833164525e-07, | |
| "loss": 2.996399688720703, | |
| "step": 850 | |
| }, | |
| { | |
| "epoch": 0.07818181818181819, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.850286353738032e-07, | |
| "loss": 2.7943418502807615, | |
| "step": 860 | |
| }, | |
| { | |
| "epoch": 0.07909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.846798312018146e-07, | |
| "loss": 2.3392452239990233, | |
| "step": 870 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.843270736455752e-07, | |
| "loss": 2.653145980834961, | |
| "step": 880 | |
| }, | |
| { | |
| "epoch": 0.0809090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.839703655824194e-07, | |
| "loss": 2.9171478271484377, | |
| "step": 890 | |
| }, | |
| { | |
| "epoch": 0.08181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.836097099219057e-07, | |
| "loss": 2.7383825302124025, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.08181818181818182, | |
| "eval_loss": 2.7882080078125, | |
| "eval_runtime": 25.34, | |
| "eval_samples_per_second": 5.801, | |
| "eval_steps_per_second": 1.46, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.08272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.832451096057912e-07, | |
| "loss": 2.408688545227051, | |
| "step": 910 | |
| }, | |
| { | |
| "epoch": 0.08363636363636363, | |
| "grad_norm": 1.1698251764699496e+19, | |
| "learning_rate": 9.82876567608008e-07, | |
| "loss": 2.5717914581298826, | |
| "step": 920 | |
| }, | |
| { | |
| "epoch": 0.08454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.825040869346404e-07, | |
| "loss": 2.71414737701416, | |
| "step": 930 | |
| }, | |
| { | |
| "epoch": 0.08545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.821276706238985e-07, | |
| "loss": 2.639026069641113, | |
| "step": 940 | |
| }, | |
| { | |
| "epoch": 0.08636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.81747321746094e-07, | |
| "loss": 2.969831848144531, | |
| "step": 950 | |
| }, | |
| { | |
| "epoch": 0.08727272727272728, | |
| "grad_norm": 4.923415706843742e+18, | |
| "learning_rate": 9.81363043403616e-07, | |
| "loss": 2.8209333419799805, | |
| "step": 960 | |
| }, | |
| { | |
| "epoch": 0.08818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.809748387309047e-07, | |
| "loss": 2.7035985946655274, | |
| "step": 970 | |
| }, | |
| { | |
| "epoch": 0.0890909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.805827108944259e-07, | |
| "loss": 3.101240348815918, | |
| "step": 980 | |
| }, | |
| { | |
| "epoch": 0.09, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.801866630926459e-07, | |
| "loss": 3.0513349533081056, | |
| "step": 990 | |
| }, | |
| { | |
| "epoch": 0.09090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.797866985560049e-07, | |
| "loss": 3.109476852416992, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.09090909090909091, | |
| "eval_loss": 2.7881357669830322, | |
| "eval_runtime": 25.2713, | |
| "eval_samples_per_second": 5.817, | |
| "eval_steps_per_second": 1.464, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.09181818181818181, | |
| "grad_norm": 8.761301725526098e+18, | |
| "learning_rate": 9.793828205468902e-07, | |
| "loss": 2.617064666748047, | |
| "step": 1010 | |
| }, | |
| { | |
| "epoch": 0.09272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.789750323596107e-07, | |
| "loss": 2.809076118469238, | |
| "step": 1020 | |
| }, | |
| { | |
| "epoch": 0.09363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.78563337320369e-07, | |
| "loss": 2.4070892333984375, | |
| "step": 1030 | |
| }, | |
| { | |
| "epoch": 0.09454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.781477387872351e-07, | |
| "loss": 2.7776512145996093, | |
| "step": 1040 | |
| }, | |
| { | |
| "epoch": 0.09545454545454546, | |
| "grad_norm": 1.8563528976131686e+18, | |
| "learning_rate": 9.77728240150118e-07, | |
| "loss": 2.613321876525879, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 0.09636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.773048448307396e-07, | |
| "loss": 2.667472076416016, | |
| "step": 1060 | |
| }, | |
| { | |
| "epoch": 0.09727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.768775562826047e-07, | |
| "loss": 2.8902267456054687, | |
| "step": 1070 | |
| }, | |
| { | |
| "epoch": 0.09818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.764463779909748e-07, | |
| "loss": 3.081578826904297, | |
| "step": 1080 | |
| }, | |
| { | |
| "epoch": 0.09909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.760113134728383e-07, | |
| "loss": 2.976548194885254, | |
| "step": 1090 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.755723662768827e-07, | |
| "loss": 2.6694435119628905, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "eval_loss": 2.7883572578430176, | |
| "eval_runtime": 25.3715, | |
| "eval_samples_per_second": 5.794, | |
| "eval_steps_per_second": 1.458, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.1009090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.751295399834653e-07, | |
| "loss": 3.10753173828125, | |
| "step": 1110 | |
| }, | |
| { | |
| "epoch": 0.10181818181818182, | |
| "grad_norm": 1.1276669319796228e+19, | |
| "learning_rate": 9.746828382045843e-07, | |
| "loss": 3.002843475341797, | |
| "step": 1120 | |
| }, | |
| { | |
| "epoch": 0.10272727272727272, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.742322645838476e-07, | |
| "loss": 3.1718584060668946, | |
| "step": 1130 | |
| }, | |
| { | |
| "epoch": 0.10363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.737778227964468e-07, | |
| "loss": 2.706940460205078, | |
| "step": 1140 | |
| }, | |
| { | |
| "epoch": 0.10454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.73319516549123e-07, | |
| "loss": 3.022201156616211, | |
| "step": 1150 | |
| }, | |
| { | |
| "epoch": 0.10545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.728573495801392e-07, | |
| "loss": 2.8677652359008787, | |
| "step": 1160 | |
| }, | |
| { | |
| "epoch": 0.10636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.723913256592495e-07, | |
| "loss": 2.7335657119750976, | |
| "step": 1170 | |
| }, | |
| { | |
| "epoch": 0.10727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.719214485876675e-07, | |
| "loss": 2.5799468994140624, | |
| "step": 1180 | |
| }, | |
| { | |
| "epoch": 0.10818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.71447722198036e-07, | |
| "loss": 2.8072345733642576, | |
| "step": 1190 | |
| }, | |
| { | |
| "epoch": 0.10909090909090909, | |
| "grad_norm": 4.2832869091002286e+18, | |
| "learning_rate": 9.709701503543952e-07, | |
| "loss": 2.8874731063842773, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.10909090909090909, | |
| "eval_loss": 2.788444757461548, | |
| "eval_runtime": 25.4151, | |
| "eval_samples_per_second": 5.784, | |
| "eval_steps_per_second": 1.456, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.11, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.70488736952152e-07, | |
| "loss": 2.6035261154174805, | |
| "step": 1210 | |
| }, | |
| { | |
| "epoch": 0.11090909090909092, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.70003485918047e-07, | |
| "loss": 3.0014669418334963, | |
| "step": 1220 | |
| }, | |
| { | |
| "epoch": 0.11181818181818182, | |
| "grad_norm": 1.4744464122615693e+19, | |
| "learning_rate": 9.695144012101237e-07, | |
| "loss": 3.6563724517822265, | |
| "step": 1230 | |
| }, | |
| { | |
| "epoch": 0.11272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.690214868176957e-07, | |
| "loss": 3.150925636291504, | |
| "step": 1240 | |
| }, | |
| { | |
| "epoch": 0.11363636363636363, | |
| "grad_norm": 1.3018001069077168e+19, | |
| "learning_rate": 9.68524746761314e-07, | |
| "loss": 2.9263046264648436, | |
| "step": 1250 | |
| }, | |
| { | |
| "epoch": 0.11454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.680241850927343e-07, | |
| "loss": 2.8352964401245115, | |
| "step": 1260 | |
| }, | |
| { | |
| "epoch": 0.11545454545454545, | |
| "grad_norm": 3.9445782289952276e+18, | |
| "learning_rate": 9.675198058948842e-07, | |
| "loss": 2.9046945571899414, | |
| "step": 1270 | |
| }, | |
| { | |
| "epoch": 0.11636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.670116132818296e-07, | |
| "loss": 2.561818504333496, | |
| "step": 1280 | |
| }, | |
| { | |
| "epoch": 0.11727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.664996113987413e-07, | |
| "loss": 2.5337194442749023, | |
| "step": 1290 | |
| }, | |
| { | |
| "epoch": 0.11818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.659838044218612e-07, | |
| "loss": 2.7506519317626954, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.11818181818181818, | |
| "eval_loss": 2.7886366844177246, | |
| "eval_runtime": 25.2984, | |
| "eval_samples_per_second": 5.811, | |
| "eval_steps_per_second": 1.463, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.1190909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.654641965584677e-07, | |
| "loss": 2.585481643676758, | |
| "step": 1310 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.649407920468429e-07, | |
| "loss": 2.387790489196777, | |
| "step": 1320 | |
| }, | |
| { | |
| "epoch": 0.12090909090909091, | |
| "grad_norm": 8.597805445987434e+18, | |
| "learning_rate": 9.644135951562358e-07, | |
| "loss": 2.409588623046875, | |
| "step": 1330 | |
| }, | |
| { | |
| "epoch": 0.12181818181818181, | |
| "grad_norm": 1.4816577791746638e+19, | |
| "learning_rate": 9.638826101868297e-07, | |
| "loss": 2.848589324951172, | |
| "step": 1340 | |
| }, | |
| { | |
| "epoch": 0.12272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.63347841469705e-07, | |
| "loss": 2.937419891357422, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 0.12363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.62809293366806e-07, | |
| "loss": 2.8799938201904296, | |
| "step": 1360 | |
| }, | |
| { | |
| "epoch": 0.12454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.62266970270904e-07, | |
| "loss": 2.6749975204467775, | |
| "step": 1370 | |
| }, | |
| { | |
| "epoch": 0.12545454545454546, | |
| "grad_norm": 1.299667714056808e+19, | |
| "learning_rate": 9.617208766055612e-07, | |
| "loss": 2.8736942291259764, | |
| "step": 1380 | |
| }, | |
| { | |
| "epoch": 0.12636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.61171016825096e-07, | |
| "loss": 2.868055534362793, | |
| "step": 1390 | |
| }, | |
| { | |
| "epoch": 0.12727272727272726, | |
| "grad_norm": 6.926160743674937e+18, | |
| "learning_rate": 9.606173954145452e-07, | |
| "loss": 3.0848974227905273, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.12727272727272726, | |
| "eval_loss": 2.7885942459106445, | |
| "eval_runtime": 25.4457, | |
| "eval_samples_per_second": 5.777, | |
| "eval_steps_per_second": 1.454, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.12818181818181817, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.600600168896283e-07, | |
| "loss": 2.7365055084228516, | |
| "step": 1410 | |
| }, | |
| { | |
| "epoch": 0.1290909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.594988857967105e-07, | |
| "loss": 3.1326847076416016, | |
| "step": 1420 | |
| }, | |
| { | |
| "epoch": 0.13, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.589340067127657e-07, | |
| "loss": 2.2127132415771484, | |
| "step": 1430 | |
| }, | |
| { | |
| "epoch": 0.13090909090909092, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.583653842453382e-07, | |
| "loss": 2.7018590927124024, | |
| "step": 1440 | |
| }, | |
| { | |
| "epoch": 0.1318181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.577930230325072e-07, | |
| "loss": 3.1343578338623046, | |
| "step": 1450 | |
| }, | |
| { | |
| "epoch": 0.13272727272727272, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.572169277428464e-07, | |
| "loss": 2.7940521240234375, | |
| "step": 1460 | |
| }, | |
| { | |
| "epoch": 0.13363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.566371030753882e-07, | |
| "loss": 2.700839614868164, | |
| "step": 1470 | |
| }, | |
| { | |
| "epoch": 0.13454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.560535537595838e-07, | |
| "loss": 2.729141616821289, | |
| "step": 1480 | |
| }, | |
| { | |
| "epoch": 0.13545454545454547, | |
| "grad_norm": 9.33644581265526e+18, | |
| "learning_rate": 9.554662845552656e-07, | |
| "loss": 2.583717155456543, | |
| "step": 1490 | |
| }, | |
| { | |
| "epoch": 0.13636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.548753002526076e-07, | |
| "loss": 2.8652286529541016, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.13636363636363635, | |
| "eval_loss": 2.7888333797454834, | |
| "eval_runtime": 25.5103, | |
| "eval_samples_per_second": 5.762, | |
| "eval_steps_per_second": 1.45, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.13727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.54280605672087e-07, | |
| "loss": 2.878238868713379, | |
| "step": 1510 | |
| }, | |
| { | |
| "epoch": 0.13818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.536822056644444e-07, | |
| "loss": 2.555144500732422, | |
| "step": 1520 | |
| }, | |
| { | |
| "epoch": 0.1390909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.530801051106448e-07, | |
| "loss": 2.8380672454833986, | |
| "step": 1530 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "grad_norm": 1.8191689261902725e+19, | |
| "learning_rate": 9.52474308921837e-07, | |
| "loss": 2.9135501861572264, | |
| "step": 1540 | |
| }, | |
| { | |
| "epoch": 0.1409090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.518648220393142e-07, | |
| "loss": 2.995115280151367, | |
| "step": 1550 | |
| }, | |
| { | |
| "epoch": 0.14181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.512516494344732e-07, | |
| "loss": 2.3885705947875975, | |
| "step": 1560 | |
| }, | |
| { | |
| "epoch": 0.14272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.506347961087743e-07, | |
| "loss": 3.084286689758301, | |
| "step": 1570 | |
| }, | |
| { | |
| "epoch": 0.14363636363636365, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.500142670937006e-07, | |
| "loss": 2.691249656677246, | |
| "step": 1580 | |
| }, | |
| { | |
| "epoch": 0.14454545454545453, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.493900674507158e-07, | |
| "loss": 2.8905223846435546, | |
| "step": 1590 | |
| }, | |
| { | |
| "epoch": 0.14545454545454545, | |
| "grad_norm": 1.9084444600023122e+18, | |
| "learning_rate": 9.487622022712247e-07, | |
| "loss": 2.9005517959594727, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.14545454545454545, | |
| "eval_loss": 2.7890355587005615, | |
| "eval_runtime": 25.4165, | |
| "eval_samples_per_second": 5.784, | |
| "eval_steps_per_second": 1.456, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.14636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.481306766765304e-07, | |
| "loss": 2.458705520629883, | |
| "step": 1610 | |
| }, | |
| { | |
| "epoch": 0.14727272727272728, | |
| "grad_norm": 7.697958532745789e+18, | |
| "learning_rate": 9.474954958177925e-07, | |
| "loss": 2.9042383193969727, | |
| "step": 1620 | |
| }, | |
| { | |
| "epoch": 0.1481818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.468566648759864e-07, | |
| "loss": 2.9171165466308593, | |
| "step": 1630 | |
| }, | |
| { | |
| "epoch": 0.14909090909090908, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.462141890618589e-07, | |
| "loss": 2.7405115127563477, | |
| "step": 1640 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "grad_norm": 8.394250109558194e+18, | |
| "learning_rate": 9.455680736158879e-07, | |
| "loss": 2.944680404663086, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 0.1509090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.449183238082383e-07, | |
| "loss": 3.1834575653076174, | |
| "step": 1660 | |
| }, | |
| { | |
| "epoch": 0.15181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.442649449387193e-07, | |
| "loss": 2.709722137451172, | |
| "step": 1670 | |
| }, | |
| { | |
| "epoch": 0.15272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.436079423367412e-07, | |
| "loss": 2.5861513137817385, | |
| "step": 1680 | |
| }, | |
| { | |
| "epoch": 0.15363636363636363, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.429473213612722e-07, | |
| "loss": 3.1419662475585937, | |
| "step": 1690 | |
| }, | |
| { | |
| "epoch": 0.15454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.422830874007943e-07, | |
| "loss": 2.909741020202637, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 0.15454545454545454, | |
| "eval_loss": 2.789142608642578, | |
| "eval_runtime": 25.3588, | |
| "eval_samples_per_second": 5.797, | |
| "eval_steps_per_second": 1.459, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 0.15545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.416152458732595e-07, | |
| "loss": 2.784795379638672, | |
| "step": 1710 | |
| }, | |
| { | |
| "epoch": 0.15636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.409438022260456e-07, | |
| "loss": 2.862109375, | |
| "step": 1720 | |
| }, | |
| { | |
| "epoch": 0.1572727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.40268761935912e-07, | |
| "loss": 2.937261199951172, | |
| "step": 1730 | |
| }, | |
| { | |
| "epoch": 0.15818181818181817, | |
| "grad_norm": 1.2604903555405447e+19, | |
| "learning_rate": 9.395901305089544e-07, | |
| "loss": 3.0458484649658204, | |
| "step": 1740 | |
| }, | |
| { | |
| "epoch": 0.1590909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.389079134805609e-07, | |
| "loss": 2.753535270690918, | |
| "step": 1750 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.382221164153654e-07, | |
| "loss": 2.8955476760864256, | |
| "step": 1760 | |
| }, | |
| { | |
| "epoch": 0.16090909090909092, | |
| "grad_norm": 7.74396099974031e+18, | |
| "learning_rate": 9.37532744907204e-07, | |
| "loss": 2.53248291015625, | |
| "step": 1770 | |
| }, | |
| { | |
| "epoch": 0.1618181818181818, | |
| "grad_norm": 1.391111567262194e+19, | |
| "learning_rate": 9.368398045790678e-07, | |
| "loss": 2.7983562469482424, | |
| "step": 1780 | |
| }, | |
| { | |
| "epoch": 0.16272727272727272, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.361433010830577e-07, | |
| "loss": 2.625954246520996, | |
| "step": 1790 | |
| }, | |
| { | |
| "epoch": 0.16363636363636364, | |
| "grad_norm": 1.6281627557734908e+19, | |
| "learning_rate": 9.354432401003386e-07, | |
| "loss": 2.65600528717041, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.16363636363636364, | |
| "eval_loss": 2.7892098426818848, | |
| "eval_runtime": 25.3434, | |
| "eval_samples_per_second": 5.8, | |
| "eval_steps_per_second": 1.46, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.16454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.347396273410924e-07, | |
| "loss": 2.858198547363281, | |
| "step": 1810 | |
| }, | |
| { | |
| "epoch": 0.16545454545454547, | |
| "grad_norm": 1.5438372507974894e+19, | |
| "learning_rate": 9.34032468544472e-07, | |
| "loss": 2.5564403533935547, | |
| "step": 1820 | |
| }, | |
| { | |
| "epoch": 0.16636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.333217694785542e-07, | |
| "loss": 2.579339599609375, | |
| "step": 1830 | |
| }, | |
| { | |
| "epoch": 0.16727272727272727, | |
| "grad_norm": 5.160486906466664e+18, | |
| "learning_rate": 9.326075359402924e-07, | |
| "loss": 2.883788299560547, | |
| "step": 1840 | |
| }, | |
| { | |
| "epoch": 0.16818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.318897737554698e-07, | |
| "loss": 2.588374900817871, | |
| "step": 1850 | |
| }, | |
| { | |
| "epoch": 0.1690909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.31168488778652e-07, | |
| "loss": 2.779450035095215, | |
| "step": 1860 | |
| }, | |
| { | |
| "epoch": 0.17, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.304436868931382e-07, | |
| "loss": 2.9528104782104494, | |
| "step": 1870 | |
| }, | |
| { | |
| "epoch": 0.1709090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.297153740109147e-07, | |
| "loss": 3.0303653717041015, | |
| "step": 1880 | |
| }, | |
| { | |
| "epoch": 0.17181818181818181, | |
| "grad_norm": 1.5527703429685182e+19, | |
| "learning_rate": 9.289835560726052e-07, | |
| "loss": 2.797162628173828, | |
| "step": 1890 | |
| }, | |
| { | |
| "epoch": 0.17272727272727273, | |
| "grad_norm": 1.553810151114906e+19, | |
| "learning_rate": 9.28248239047424e-07, | |
| "loss": 2.6478057861328126, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.17272727272727273, | |
| "eval_loss": 2.789252281188965, | |
| "eval_runtime": 25.3648, | |
| "eval_samples_per_second": 5.795, | |
| "eval_steps_per_second": 1.459, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.17363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.275094289331253e-07, | |
| "loss": 2.7918701171875, | |
| "step": 1910 | |
| }, | |
| { | |
| "epoch": 0.17454545454545456, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.267671317559562e-07, | |
| "loss": 2.873759078979492, | |
| "step": 1920 | |
| }, | |
| { | |
| "epoch": 0.17545454545454545, | |
| "grad_norm": 7.946095217390649e+18, | |
| "learning_rate": 9.260213535706062e-07, | |
| "loss": 3.0809797286987304, | |
| "step": 1930 | |
| }, | |
| { | |
| "epoch": 0.17636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.252721004601587e-07, | |
| "loss": 2.7745038986206056, | |
| "step": 1940 | |
| }, | |
| { | |
| "epoch": 0.17727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.245193785360406e-07, | |
| "loss": 2.998860740661621, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 0.1781818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.23763193937973e-07, | |
| "loss": 2.7716712951660156, | |
| "step": 1960 | |
| }, | |
| { | |
| "epoch": 0.17909090909090908, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.23003552833921e-07, | |
| "loss": 3.0993738174438477, | |
| "step": 1970 | |
| }, | |
| { | |
| "epoch": 0.18, | |
| "grad_norm": 1.3247786904654447e+19, | |
| "learning_rate": 9.222404614200436e-07, | |
| "loss": 2.863362121582031, | |
| "step": 1980 | |
| }, | |
| { | |
| "epoch": 0.1809090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.214739259206423e-07, | |
| "loss": 3.02794189453125, | |
| "step": 1990 | |
| }, | |
| { | |
| "epoch": 0.18181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.207039525881117e-07, | |
| "loss": 3.025308036804199, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.18181818181818182, | |
| "eval_loss": 2.7894153594970703, | |
| "eval_runtime": 25.359, | |
| "eval_samples_per_second": 5.797, | |
| "eval_steps_per_second": 1.459, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.18272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.199305477028869e-07, | |
| "loss": 2.7843177795410154, | |
| "step": 2010 | |
| }, | |
| { | |
| "epoch": 0.18363636363636363, | |
| "grad_norm": 5.96282857880132e+18, | |
| "learning_rate": 9.19153717573394e-07, | |
| "loss": 2.802363967895508, | |
| "step": 2020 | |
| }, | |
| { | |
| "epoch": 0.18454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.183734685359972e-07, | |
| "loss": 3.1789117813110352, | |
| "step": 2030 | |
| }, | |
| { | |
| "epoch": 0.18545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.175898069549476e-07, | |
| "loss": 2.3378616333007813, | |
| "step": 2040 | |
| }, | |
| { | |
| "epoch": 0.18636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.168027392223319e-07, | |
| "loss": 3.002253532409668, | |
| "step": 2050 | |
| }, | |
| { | |
| "epoch": 0.18727272727272729, | |
| "grad_norm": 9.47185286863913e+18, | |
| "learning_rate": 9.160122717580194e-07, | |
| "loss": 2.89359073638916, | |
| "step": 2060 | |
| }, | |
| { | |
| "epoch": 0.18818181818181817, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.152184110096097e-07, | |
| "loss": 2.685699462890625, | |
| "step": 2070 | |
| }, | |
| { | |
| "epoch": 0.1890909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.144211634523808e-07, | |
| "loss": 3.139577293395996, | |
| "step": 2080 | |
| }, | |
| { | |
| "epoch": 0.19, | |
| "grad_norm": 6.611679527410074e+18, | |
| "learning_rate": 9.13620535589236e-07, | |
| "loss": 2.7923168182373046, | |
| "step": 2090 | |
| }, | |
| { | |
| "epoch": 0.19090909090909092, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.128165339506502e-07, | |
| "loss": 2.4308433532714844, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 0.19090909090909092, | |
| "eval_loss": 2.789486885070801, | |
| "eval_runtime": 25.465, | |
| "eval_samples_per_second": 5.773, | |
| "eval_steps_per_second": 1.453, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 0.1918181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.120091650946171e-07, | |
| "loss": 3.2387130737304686, | |
| "step": 2110 | |
| }, | |
| { | |
| "epoch": 0.19272727272727272, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.111984356065966e-07, | |
| "loss": 3.4744667053222655, | |
| "step": 2120 | |
| }, | |
| { | |
| "epoch": 0.19363636363636363, | |
| "grad_norm": 1.4776962387797869e+19, | |
| "learning_rate": 9.103843520994592e-07, | |
| "loss": 3.85797233581543, | |
| "step": 2130 | |
| }, | |
| { | |
| "epoch": 0.19454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.095669212134339e-07, | |
| "loss": 3.818370819091797, | |
| "step": 2140 | |
| }, | |
| { | |
| "epoch": 0.19545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.087461496160528e-07, | |
| "loss": 3.4372634887695312, | |
| "step": 2150 | |
| }, | |
| { | |
| "epoch": 0.19636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.079220440020972e-07, | |
| "loss": 2.8101428985595702, | |
| "step": 2160 | |
| }, | |
| { | |
| "epoch": 0.19727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.070946110935431e-07, | |
| "loss": 2.7843332290649414, | |
| "step": 2170 | |
| }, | |
| { | |
| "epoch": 0.19818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.062638576395061e-07, | |
| "loss": 2.402296447753906, | |
| "step": 2180 | |
| }, | |
| { | |
| "epoch": 0.1990909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.054297904161867e-07, | |
| "loss": 2.8248491287231445, | |
| "step": 2190 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.045924162268144e-07, | |
| "loss": 2.7025005340576174, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "eval_loss": 2.7895781993865967, | |
| "eval_runtime": 25.4794, | |
| "eval_samples_per_second": 5.769, | |
| "eval_steps_per_second": 1.452, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 0.2009090909090909, | |
| "grad_norm": 1.249766048976706e+19, | |
| "learning_rate": 9.037517419015928e-07, | |
| "loss": 2.798247528076172, | |
| "step": 2210 | |
| }, | |
| { | |
| "epoch": 0.2018181818181818, | |
| "grad_norm": 1.637043731093363e+19, | |
| "learning_rate": 9.029077742976437e-07, | |
| "loss": 2.894302177429199, | |
| "step": 2220 | |
| }, | |
| { | |
| "epoch": 0.20272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.020605202989513e-07, | |
| "loss": 2.6775253295898436, | |
| "step": 2230 | |
| }, | |
| { | |
| "epoch": 0.20363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 9.012099868163056e-07, | |
| "loss": 2.829499053955078, | |
| "step": 2240 | |
| }, | |
| { | |
| "epoch": 0.20454545454545456, | |
| "grad_norm": 1.7267504658780258e+19, | |
| "learning_rate": 9.003561807872469e-07, | |
| "loss": 2.526635932922363, | |
| "step": 2250 | |
| }, | |
| { | |
| "epoch": 0.20545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.994991091760076e-07, | |
| "loss": 2.897632598876953, | |
| "step": 2260 | |
| }, | |
| { | |
| "epoch": 0.20636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.986387789734576e-07, | |
| "loss": 2.66369514465332, | |
| "step": 2270 | |
| }, | |
| { | |
| "epoch": 0.20727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.977751971970456e-07, | |
| "loss": 2.954084587097168, | |
| "step": 2280 | |
| }, | |
| { | |
| "epoch": 0.2081818181818182, | |
| "grad_norm": 4.468861107246727e+18, | |
| "learning_rate": 8.969083708907424e-07, | |
| "loss": 3.04129581451416, | |
| "step": 2290 | |
| }, | |
| { | |
| "epoch": 0.20909090909090908, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.960383071249835e-07, | |
| "loss": 2.8931554794311523, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 0.20909090909090908, | |
| "eval_loss": 2.78959321975708, | |
| "eval_runtime": 25.2525, | |
| "eval_samples_per_second": 5.821, | |
| "eval_steps_per_second": 1.465, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 0.21, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.951650129966112e-07, | |
| "loss": 2.716269683837891, | |
| "step": 2310 | |
| }, | |
| { | |
| "epoch": 0.2109090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.942884956288171e-07, | |
| "loss": 2.980978012084961, | |
| "step": 2320 | |
| }, | |
| { | |
| "epoch": 0.21181818181818182, | |
| "grad_norm": 6.872592537169691e+18, | |
| "learning_rate": 8.934087621710835e-07, | |
| "loss": 2.609796905517578, | |
| "step": 2330 | |
| }, | |
| { | |
| "epoch": 0.21272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.925258197991258e-07, | |
| "loss": 3.0382633209228516, | |
| "step": 2340 | |
| }, | |
| { | |
| "epoch": 0.21363636363636362, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.91639675714833e-07, | |
| "loss": 3.127687454223633, | |
| "step": 2350 | |
| }, | |
| { | |
| "epoch": 0.21454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.907503371462099e-07, | |
| "loss": 2.8488794326782227, | |
| "step": 2360 | |
| }, | |
| { | |
| "epoch": 0.21545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.898578113473176e-07, | |
| "loss": 2.7067138671875, | |
| "step": 2370 | |
| }, | |
| { | |
| "epoch": 0.21636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.889621055982144e-07, | |
| "loss": 2.789401626586914, | |
| "step": 2380 | |
| }, | |
| { | |
| "epoch": 0.21727272727272728, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.880632272048962e-07, | |
| "loss": 3.3033519744873048, | |
| "step": 2390 | |
| }, | |
| { | |
| "epoch": 0.21818181818181817, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.871611834992379e-07, | |
| "loss": 2.878365707397461, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 0.21818181818181817, | |
| "eval_loss": 2.789731025695801, | |
| "eval_runtime": 25.3629, | |
| "eval_samples_per_second": 5.796, | |
| "eval_steps_per_second": 1.459, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 0.2190909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.862559818389321e-07, | |
| "loss": 2.9361312866210936, | |
| "step": 2410 | |
| }, | |
| { | |
| "epoch": 0.22, | |
| "grad_norm": 1.4403865107144638e+19, | |
| "learning_rate": 8.853476296074305e-07, | |
| "loss": 2.718800735473633, | |
| "step": 2420 | |
| }, | |
| { | |
| "epoch": 0.22090909090909092, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.844361342138827e-07, | |
| "loss": 2.7465553283691406, | |
| "step": 2430 | |
| }, | |
| { | |
| "epoch": 0.22181818181818183, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.835215030930761e-07, | |
| "loss": 2.5267559051513673, | |
| "step": 2440 | |
| }, | |
| { | |
| "epoch": 0.22272727272727272, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.82603743705375e-07, | |
| "loss": 2.8523298263549806, | |
| "step": 2450 | |
| }, | |
| { | |
| "epoch": 0.22363636363636363, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.81682863536661e-07, | |
| "loss": 2.890406608581543, | |
| "step": 2460 | |
| }, | |
| { | |
| "epoch": 0.22454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.807588700982699e-07, | |
| "loss": 3.1316598892211913, | |
| "step": 2470 | |
| }, | |
| { | |
| "epoch": 0.22545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.798317709269318e-07, | |
| "loss": 2.738657760620117, | |
| "step": 2480 | |
| }, | |
| { | |
| "epoch": 0.22636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.789015735847098e-07, | |
| "loss": 2.8348062515258787, | |
| "step": 2490 | |
| }, | |
| { | |
| "epoch": 0.22727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.779682856589367e-07, | |
| "loss": 3.0129791259765626, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 0.22727272727272727, | |
| "eval_loss": 2.7899017333984375, | |
| "eval_runtime": 25.3362, | |
| "eval_samples_per_second": 5.802, | |
| "eval_steps_per_second": 1.46, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 0.22818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.770319147621557e-07, | |
| "loss": 2.794989013671875, | |
| "step": 2510 | |
| }, | |
| { | |
| "epoch": 0.2290909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.760924685320557e-07, | |
| "loss": 2.7114553451538086, | |
| "step": 2520 | |
| }, | |
| { | |
| "epoch": 0.23, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.751499546314105e-07, | |
| "loss": 2.5232223510742187, | |
| "step": 2530 | |
| }, | |
| { | |
| "epoch": 0.2309090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.742043807480162e-07, | |
| "loss": 2.8785938262939452, | |
| "step": 2540 | |
| }, | |
| { | |
| "epoch": 0.2318181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.732557545946278e-07, | |
| "loss": 2.575123596191406, | |
| "step": 2550 | |
| }, | |
| { | |
| "epoch": 0.23272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.723040839088969e-07, | |
| "loss": 2.522659683227539, | |
| "step": 2560 | |
| }, | |
| { | |
| "epoch": 0.23363636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.713493764533088e-07, | |
| "loss": 2.8160863876342774, | |
| "step": 2570 | |
| }, | |
| { | |
| "epoch": 0.23454545454545456, | |
| "grad_norm": 1.7132498099906806e+18, | |
| "learning_rate": 8.703916400151183e-07, | |
| "loss": 2.9550497055053713, | |
| "step": 2580 | |
| }, | |
| { | |
| "epoch": 0.23545454545454544, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.694308824062867e-07, | |
| "loss": 2.701577568054199, | |
| "step": 2590 | |
| }, | |
| { | |
| "epoch": 0.23636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.684671114634183e-07, | |
| "loss": 2.701395606994629, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 0.23636363636363636, | |
| "eval_loss": 2.7901124954223633, | |
| "eval_runtime": 25.2085, | |
| "eval_samples_per_second": 5.831, | |
| "eval_steps_per_second": 1.468, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 0.23727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.675003350476961e-07, | |
| "loss": 3.045684814453125, | |
| "step": 2610 | |
| }, | |
| { | |
| "epoch": 0.2381818181818182, | |
| "grad_norm": 1.7417936541157687e+19, | |
| "learning_rate": 8.66530561044818e-07, | |
| "loss": 3.0101932525634765, | |
| "step": 2620 | |
| }, | |
| { | |
| "epoch": 0.2390909090909091, | |
| "grad_norm": 1.790928409835497e+19, | |
| "learning_rate": 8.655577973649321e-07, | |
| "loss": 2.312086486816406, | |
| "step": 2630 | |
| }, | |
| { | |
| "epoch": 0.24, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.645820519425722e-07, | |
| "loss": 3.022278594970703, | |
| "step": 2640 | |
| }, | |
| { | |
| "epoch": 0.2409090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.636033327365937e-07, | |
| "loss": 2.964511489868164, | |
| "step": 2650 | |
| }, | |
| { | |
| "epoch": 0.24181818181818182, | |
| "grad_norm": 6.040087962350256e+18, | |
| "learning_rate": 8.626216477301081e-07, | |
| "loss": 3.050629997253418, | |
| "step": 2660 | |
| }, | |
| { | |
| "epoch": 0.24272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.616370049304175e-07, | |
| "loss": 2.869434356689453, | |
| "step": 2670 | |
| }, | |
| { | |
| "epoch": 0.24363636363636362, | |
| "grad_norm": 1.4910112146409914e+19, | |
| "learning_rate": 8.606494123689508e-07, | |
| "loss": 2.7562395095825196, | |
| "step": 2680 | |
| }, | |
| { | |
| "epoch": 0.24454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.596588781011963e-07, | |
| "loss": 2.5833467483520507, | |
| "step": 2690 | |
| }, | |
| { | |
| "epoch": 0.24545454545454545, | |
| "grad_norm": 1.4346714691756098e+19, | |
| "learning_rate": 8.586654102066374e-07, | |
| "loss": 2.8177486419677735, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 0.24545454545454545, | |
| "eval_loss": 2.790109157562256, | |
| "eval_runtime": 25.304, | |
| "eval_samples_per_second": 5.809, | |
| "eval_steps_per_second": 1.462, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 0.24636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.576690167886859e-07, | |
| "loss": 2.540431785583496, | |
| "step": 2710 | |
| }, | |
| { | |
| "epoch": 0.24727272727272728, | |
| "grad_norm": 2.7262607964252406e+18, | |
| "learning_rate": 8.566697059746166e-07, | |
| "loss": 2.6125823974609377, | |
| "step": 2720 | |
| }, | |
| { | |
| "epoch": 0.24818181818181817, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.556674859155e-07, | |
| "loss": 2.758362579345703, | |
| "step": 2730 | |
| }, | |
| { | |
| "epoch": 0.24909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.546623647861369e-07, | |
| "loss": 3.0073740005493166, | |
| "step": 2740 | |
| }, | |
| { | |
| "epoch": 0.25, | |
| "grad_norm": 7.052120795752956e+18, | |
| "learning_rate": 8.536543507849911e-07, | |
| "loss": 2.758180618286133, | |
| "step": 2750 | |
| }, | |
| { | |
| "epoch": 0.2509090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.526434521341226e-07, | |
| "loss": 2.963096046447754, | |
| "step": 2760 | |
| }, | |
| { | |
| "epoch": 0.25181818181818183, | |
| "grad_norm": 1.3421638384703504e+19, | |
| "learning_rate": 8.516296770791207e-07, | |
| "loss": 2.8563003540039062, | |
| "step": 2770 | |
| }, | |
| { | |
| "epoch": 0.25272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.506130338890365e-07, | |
| "loss": 3.0656408309936523, | |
| "step": 2780 | |
| }, | |
| { | |
| "epoch": 0.25363636363636366, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.495935308563158e-07, | |
| "loss": 2.983555221557617, | |
| "step": 2790 | |
| }, | |
| { | |
| "epoch": 0.2545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.485711762967312e-07, | |
| "loss": 2.951832580566406, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 0.2545454545454545, | |
| "eval_loss": 2.790069818496704, | |
| "eval_runtime": 25.4101, | |
| "eval_samples_per_second": 5.785, | |
| "eval_steps_per_second": 1.456, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 0.25545454545454543, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.47545978549314e-07, | |
| "loss": 2.5605377197265624, | |
| "step": 2810 | |
| }, | |
| { | |
| "epoch": 0.25636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.46517945976287e-07, | |
| "loss": 2.6139829635620115, | |
| "step": 2820 | |
| }, | |
| { | |
| "epoch": 0.25727272727272726, | |
| "grad_norm": 1.3888741710508327e+19, | |
| "learning_rate": 8.454870869629955e-07, | |
| "loss": 2.8513839721679686, | |
| "step": 2830 | |
| }, | |
| { | |
| "epoch": 0.2581818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.444534099178393e-07, | |
| "loss": 2.7712711334228515, | |
| "step": 2840 | |
| }, | |
| { | |
| "epoch": 0.2590909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.434169232722041e-07, | |
| "loss": 2.187318801879883, | |
| "step": 2850 | |
| }, | |
| { | |
| "epoch": 0.26, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.423776354803925e-07, | |
| "loss": 3.0384477615356444, | |
| "step": 2860 | |
| }, | |
| { | |
| "epoch": 0.2609090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.41335555019555e-07, | |
| "loss": 2.9354883193969727, | |
| "step": 2870 | |
| }, | |
| { | |
| "epoch": 0.26181818181818184, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.402906903896216e-07, | |
| "loss": 3.0859096527099608, | |
| "step": 2880 | |
| }, | |
| { | |
| "epoch": 0.26272727272727275, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.392430501132316e-07, | |
| "loss": 3.006024169921875, | |
| "step": 2890 | |
| }, | |
| { | |
| "epoch": 0.2636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.381926427356642e-07, | |
| "loss": 2.832804298400879, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 0.2636363636363636, | |
| "eval_loss": 2.7902262210845947, | |
| "eval_runtime": 25.2761, | |
| "eval_samples_per_second": 5.816, | |
| "eval_steps_per_second": 1.464, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 0.26454545454545453, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.37139476824769e-07, | |
| "loss": 2.8630512237548826, | |
| "step": 2910 | |
| }, | |
| { | |
| "epoch": 0.26545454545454544, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.360835609708967e-07, | |
| "loss": 2.8531957626342774, | |
| "step": 2920 | |
| }, | |
| { | |
| "epoch": 0.26636363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.35024903786828e-07, | |
| "loss": 2.766724967956543, | |
| "step": 2930 | |
| }, | |
| { | |
| "epoch": 0.2672727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.339635139077035e-07, | |
| "loss": 3.115190887451172, | |
| "step": 2940 | |
| }, | |
| { | |
| "epoch": 0.2681818181818182, | |
| "grad_norm": 1.7490984795172241e+19, | |
| "learning_rate": 8.32899399990954e-07, | |
| "loss": 3.0780994415283205, | |
| "step": 2950 | |
| }, | |
| { | |
| "epoch": 0.2690909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.318325707162293e-07, | |
| "loss": 2.6672946929931642, | |
| "step": 2960 | |
| }, | |
| { | |
| "epoch": 0.27, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.307630347853273e-07, | |
| "loss": 2.781101417541504, | |
| "step": 2970 | |
| }, | |
| { | |
| "epoch": 0.27090909090909093, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.296908009221242e-07, | |
| "loss": 2.4955059051513673, | |
| "step": 2980 | |
| }, | |
| { | |
| "epoch": 0.2718181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.286158778725011e-07, | |
| "loss": 2.4922977447509767, | |
| "step": 2990 | |
| }, | |
| { | |
| "epoch": 0.2727272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.275382744042747e-07, | |
| "loss": 2.6550668716430663, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.2727272727272727, | |
| "eval_loss": 2.7902653217315674, | |
| "eval_runtime": 25.2219, | |
| "eval_samples_per_second": 5.828, | |
| "eval_steps_per_second": 1.467, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.2736363636363636, | |
| "grad_norm": 1.523740707118488e+19, | |
| "learning_rate": 8.26457999307125e-07, | |
| "loss": 2.947774887084961, | |
| "step": 3010 | |
| }, | |
| { | |
| "epoch": 0.27454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.253750613925234e-07, | |
| "loss": 2.497574806213379, | |
| "step": 3020 | |
| }, | |
| { | |
| "epoch": 0.27545454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.242894694936614e-07, | |
| "loss": 2.8819416046142576, | |
| "step": 3030 | |
| }, | |
| { | |
| "epoch": 0.27636363636363637, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.23201232465378e-07, | |
| "loss": 2.79040641784668, | |
| "step": 3040 | |
| }, | |
| { | |
| "epoch": 0.2772727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.221103591840881e-07, | |
| "loss": 2.8814857482910154, | |
| "step": 3050 | |
| }, | |
| { | |
| "epoch": 0.2781818181818182, | |
| "grad_norm": 7.556027524518511e+18, | |
| "learning_rate": 8.210168585477091e-07, | |
| "loss": 2.807586669921875, | |
| "step": 3060 | |
| }, | |
| { | |
| "epoch": 0.2790909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.199207394755891e-07, | |
| "loss": 2.8781753540039063, | |
| "step": 3070 | |
| }, | |
| { | |
| "epoch": 0.28, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.188220109084346e-07, | |
| "loss": 2.8258827209472654, | |
| "step": 3080 | |
| }, | |
| { | |
| "epoch": 0.2809090909090909, | |
| "grad_norm": 7.713179072209093e+18, | |
| "learning_rate": 8.17720681808236e-07, | |
| "loss": 2.972319221496582, | |
| "step": 3090 | |
| }, | |
| { | |
| "epoch": 0.2818181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.16616761158196e-07, | |
| "loss": 2.79774169921875, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 0.2818181818181818, | |
| "eval_loss": 2.7904176712036133, | |
| "eval_runtime": 25.3955, | |
| "eval_samples_per_second": 5.788, | |
| "eval_steps_per_second": 1.457, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 0.2827272727272727, | |
| "grad_norm": 6.054117180964864e+18, | |
| "learning_rate": 8.155102579626558e-07, | |
| "loss": 2.4805845260620116, | |
| "step": 3110 | |
| }, | |
| { | |
| "epoch": 0.28363636363636363, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.144011812470215e-07, | |
| "loss": 2.973079299926758, | |
| "step": 3120 | |
| }, | |
| { | |
| "epoch": 0.28454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.132895400576904e-07, | |
| "loss": 2.4612403869628907, | |
| "step": 3130 | |
| }, | |
| { | |
| "epoch": 0.28545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.121753434619778e-07, | |
| "loss": 2.7316892623901365, | |
| "step": 3140 | |
| }, | |
| { | |
| "epoch": 0.2863636363636364, | |
| "grad_norm": 1.4829298041768378e+19, | |
| "learning_rate": 8.110586005480426e-07, | |
| "loss": 2.7744102478027344, | |
| "step": 3150 | |
| }, | |
| { | |
| "epoch": 0.2872727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.09939320424813e-07, | |
| "loss": 3.0601181030273437, | |
| "step": 3160 | |
| }, | |
| { | |
| "epoch": 0.2881818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.088175122219127e-07, | |
| "loss": 3.284495544433594, | |
| "step": 3170 | |
| }, | |
| { | |
| "epoch": 0.28909090909090907, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.076931850895858e-07, | |
| "loss": 3.0675275802612303, | |
| "step": 3180 | |
| }, | |
| { | |
| "epoch": 0.29, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.065663481986229e-07, | |
| "loss": 2.9712600708007812, | |
| "step": 3190 | |
| }, | |
| { | |
| "epoch": 0.2909090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.054370107402858e-07, | |
| "loss": 3.011086654663086, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 0.2909090909090909, | |
| "eval_loss": 2.7904467582702637, | |
| "eval_runtime": 25.4518, | |
| "eval_samples_per_second": 5.776, | |
| "eval_steps_per_second": 1.454, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 0.2918181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.043051819262327e-07, | |
| "loss": 2.8368200302124023, | |
| "step": 3210 | |
| }, | |
| { | |
| "epoch": 0.2927272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 8.031708709884429e-07, | |
| "loss": 3.306093215942383, | |
| "step": 3220 | |
| }, | |
| { | |
| "epoch": 0.29363636363636364, | |
| "grad_norm": 6.184249329914479e+18, | |
| "learning_rate": 8.020340871791417e-07, | |
| "loss": 2.9534244537353516, | |
| "step": 3230 | |
| }, | |
| { | |
| "epoch": 0.29454545454545455, | |
| "grad_norm": 1.056424515862528e+19, | |
| "learning_rate": 8.008948397707252e-07, | |
| "loss": 2.3959999084472656, | |
| "step": 3240 | |
| }, | |
| { | |
| "epoch": 0.29545454545454547, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.997531380556833e-07, | |
| "loss": 2.98052921295166, | |
| "step": 3250 | |
| }, | |
| { | |
| "epoch": 0.2963636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.986089913465261e-07, | |
| "loss": 2.84487361907959, | |
| "step": 3260 | |
| }, | |
| { | |
| "epoch": 0.2972727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.974624089757064e-07, | |
| "loss": 2.6605175018310545, | |
| "step": 3270 | |
| }, | |
| { | |
| "epoch": 0.29818181818181816, | |
| "grad_norm": 7.253827303380091e+18, | |
| "learning_rate": 7.963134002955431e-07, | |
| "loss": 2.5641525268554686, | |
| "step": 3280 | |
| }, | |
| { | |
| "epoch": 0.2990909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.951619746781472e-07, | |
| "loss": 2.7621686935424803, | |
| "step": 3290 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.940081415153428e-07, | |
| "loss": 2.810822296142578, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "eval_loss": 2.7905690670013428, | |
| "eval_runtime": 25.3785, | |
| "eval_samples_per_second": 5.792, | |
| "eval_steps_per_second": 1.458, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 0.3009090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.928519102185922e-07, | |
| "loss": 2.5717878341674805, | |
| "step": 3310 | |
| }, | |
| { | |
| "epoch": 0.3018181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.91693290218918e-07, | |
| "loss": 2.659162902832031, | |
| "step": 3320 | |
| }, | |
| { | |
| "epoch": 0.30272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.905322909668272e-07, | |
| "loss": 2.7497512817382814, | |
| "step": 3330 | |
| }, | |
| { | |
| "epoch": 0.30363636363636365, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.893689219322336e-07, | |
| "loss": 3.2085891723632813, | |
| "step": 3340 | |
| }, | |
| { | |
| "epoch": 0.30454545454545456, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.882031926043803e-07, | |
| "loss": 2.8847372055053713, | |
| "step": 3350 | |
| }, | |
| { | |
| "epoch": 0.3054545454545455, | |
| "grad_norm": 7.063715695623668e+18, | |
| "learning_rate": 7.870351124917629e-07, | |
| "loss": 2.7339805603027343, | |
| "step": 3360 | |
| }, | |
| { | |
| "epoch": 0.30636363636363634, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.858646911220512e-07, | |
| "loss": 2.932914924621582, | |
| "step": 3370 | |
| }, | |
| { | |
| "epoch": 0.30727272727272725, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.846919380420125e-07, | |
| "loss": 3.179635238647461, | |
| "step": 3380 | |
| }, | |
| { | |
| "epoch": 0.30818181818181817, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.835168628174325e-07, | |
| "loss": 2.679301071166992, | |
| "step": 3390 | |
| }, | |
| { | |
| "epoch": 0.3090909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.823394750330386e-07, | |
| "loss": 2.8976104736328123, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 0.3090909090909091, | |
| "eval_loss": 2.7906525135040283, | |
| "eval_runtime": 25.3974, | |
| "eval_samples_per_second": 5.788, | |
| "eval_steps_per_second": 1.457, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 0.31, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.811597842924203e-07, | |
| "loss": 3.08035888671875, | |
| "step": 3410 | |
| }, | |
| { | |
| "epoch": 0.3109090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.799778002179523e-07, | |
| "loss": 2.6392980575561524, | |
| "step": 3420 | |
| }, | |
| { | |
| "epoch": 0.3118181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.787935324507149e-07, | |
| "loss": 2.597859001159668, | |
| "step": 3430 | |
| }, | |
| { | |
| "epoch": 0.31272727272727274, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.776069906504159e-07, | |
| "loss": 2.9615325927734375, | |
| "step": 3440 | |
| }, | |
| { | |
| "epoch": 0.31363636363636366, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.764181844953116e-07, | |
| "loss": 2.8327489852905274, | |
| "step": 3450 | |
| }, | |
| { | |
| "epoch": 0.3145454545454546, | |
| "grad_norm": 9.005495011717939e+18, | |
| "learning_rate": 7.752271236821282e-07, | |
| "loss": 2.8600826263427734, | |
| "step": 3460 | |
| }, | |
| { | |
| "epoch": 0.31545454545454543, | |
| "grad_norm": 4.708173111567057e+18, | |
| "learning_rate": 7.74033817925982e-07, | |
| "loss": 2.58587646484375, | |
| "step": 3470 | |
| }, | |
| { | |
| "epoch": 0.31636363636363635, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.728382769603011e-07, | |
| "loss": 2.7968507766723634, | |
| "step": 3480 | |
| }, | |
| { | |
| "epoch": 0.31727272727272726, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.716405105367452e-07, | |
| "loss": 2.4269617080688475, | |
| "step": 3490 | |
| }, | |
| { | |
| "epoch": 0.3181818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.704405284251267e-07, | |
| "loss": 2.9382980346679686, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 0.3181818181818182, | |
| "eval_loss": 2.790801763534546, | |
| "eval_runtime": 25.2758, | |
| "eval_samples_per_second": 5.816, | |
| "eval_steps_per_second": 1.464, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 0.3190909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.6923834041333e-07, | |
| "loss": 3.010089111328125, | |
| "step": 3510 | |
| }, | |
| { | |
| "epoch": 0.32, | |
| "grad_norm": 1.7403037059089695e+19, | |
| "learning_rate": 7.680339563072333e-07, | |
| "loss": 3.1997365951538086, | |
| "step": 3520 | |
| }, | |
| { | |
| "epoch": 0.3209090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.668273859306269e-07, | |
| "loss": 2.353929138183594, | |
| "step": 3530 | |
| }, | |
| { | |
| "epoch": 0.32181818181818184, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.656186391251341e-07, | |
| "loss": 2.7316473007202147, | |
| "step": 3540 | |
| }, | |
| { | |
| "epoch": 0.32272727272727275, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.644077257501308e-07, | |
| "loss": 2.843223762512207, | |
| "step": 3550 | |
| }, | |
| { | |
| "epoch": 0.3236363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.631946556826647e-07, | |
| "loss": 2.877858543395996, | |
| "step": 3560 | |
| }, | |
| { | |
| "epoch": 0.3245454545454545, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.619794388173751e-07, | |
| "loss": 2.1713531494140623, | |
| "step": 3570 | |
| }, | |
| { | |
| "epoch": 0.32545454545454544, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.60762085066412e-07, | |
| "loss": 2.820762825012207, | |
| "step": 3580 | |
| }, | |
| { | |
| "epoch": 0.32636363636363636, | |
| "grad_norm": 1.3926951938596798e+19, | |
| "learning_rate": 7.595426043593556e-07, | |
| "loss": 2.7209442138671873, | |
| "step": 3590 | |
| }, | |
| { | |
| "epoch": 0.32727272727272727, | |
| "grad_norm": 9.029549577354609e+18, | |
| "learning_rate": 7.583210066431346e-07, | |
| "loss": 2.9243946075439453, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 0.32727272727272727, | |
| "eval_loss": 2.790870189666748, | |
| "eval_runtime": 25.1317, | |
| "eval_samples_per_second": 5.849, | |
| "eval_steps_per_second": 1.472, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 0.3281818181818182, | |
| "grad_norm": 7.039327015891108e+17, | |
| "learning_rate": 7.570973018819458e-07, | |
| "loss": 2.6024728775024415, | |
| "step": 3610 | |
| }, | |
| { | |
| "epoch": 0.3290909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.558715000571725e-07, | |
| "loss": 2.9454456329345704, | |
| "step": 3620 | |
| }, | |
| { | |
| "epoch": 0.33, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.546436111673027e-07, | |
| "loss": 2.9420135498046873, | |
| "step": 3630 | |
| }, | |
| { | |
| "epoch": 0.33090909090909093, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.534136452278486e-07, | |
| "loss": 2.928874206542969, | |
| "step": 3640 | |
| }, | |
| { | |
| "epoch": 0.33181818181818185, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.521816122712637e-07, | |
| "loss": 2.8279190063476562, | |
| "step": 3650 | |
| }, | |
| { | |
| "epoch": 0.3327272727272727, | |
| "grad_norm": 1.0277020835612983e+19, | |
| "learning_rate": 7.509475223468618e-07, | |
| "loss": 2.930580902099609, | |
| "step": 3660 | |
| }, | |
| { | |
| "epoch": 0.3336363636363636, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.497113855207347e-07, | |
| "loss": 2.815085220336914, | |
| "step": 3670 | |
| }, | |
| { | |
| "epoch": 0.33454545454545453, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.4847321187567e-07, | |
| "loss": 2.9397621154785156, | |
| "step": 3680 | |
| }, | |
| { | |
| "epoch": 0.33545454545454545, | |
| "grad_norm": 1.3450457683979141e+19, | |
| "learning_rate": 7.472330115110696e-07, | |
| "loss": 2.5939712524414062, | |
| "step": 3690 | |
| }, | |
| { | |
| "epoch": 0.33636363636363636, | |
| "grad_norm": 1.2687692382930469e+19, | |
| "learning_rate": 7.459907945428657e-07, | |
| "loss": 3.0160362243652346, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 0.33636363636363636, | |
| "eval_loss": 2.7909200191497803, | |
| "eval_runtime": 25.1818, | |
| "eval_samples_per_second": 5.838, | |
| "eval_steps_per_second": 1.469, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 0.3372727272727273, | |
| "grad_norm": 1.4489552247320478e+19, | |
| "learning_rate": 7.447465711034404e-07, | |
| "loss": 2.867232894897461, | |
| "step": 3710 | |
| }, | |
| { | |
| "epoch": 0.3381818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.43500351341541e-07, | |
| "loss": 2.8905174255371096, | |
| "step": 3720 | |
| }, | |
| { | |
| "epoch": 0.3390909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.422521454221989e-07, | |
| "loss": 2.171013069152832, | |
| "step": 3730 | |
| }, | |
| { | |
| "epoch": 0.34, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.410019635266459e-07, | |
| "loss": 2.721842384338379, | |
| "step": 3740 | |
| }, | |
| { | |
| "epoch": 0.3409090909090909, | |
| "grad_norm": 3.9303648421829673e+18, | |
| "learning_rate": 7.397498158522306e-07, | |
| "loss": 2.7071441650390624, | |
| "step": 3750 | |
| }, | |
| { | |
| "epoch": 0.3418181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.384957126123365e-07, | |
| "loss": 3.0133041381835937, | |
| "step": 3760 | |
| }, | |
| { | |
| "epoch": 0.3427272727272727, | |
| "grad_norm": 7.447019192961729e+18, | |
| "learning_rate": 7.37239664036298e-07, | |
| "loss": 2.8034151077270506, | |
| "step": 3770 | |
| }, | |
| { | |
| "epoch": 0.34363636363636363, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.359816803693166e-07, | |
| "loss": 2.6570461273193358, | |
| "step": 3780 | |
| }, | |
| { | |
| "epoch": 0.34454545454545454, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.347217718723783e-07, | |
| "loss": 3.0319595336914062, | |
| "step": 3790 | |
| }, | |
| { | |
| "epoch": 0.34545454545454546, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.334599488221689e-07, | |
| "loss": 2.714298439025879, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 0.34545454545454546, | |
| "eval_loss": 2.7910683155059814, | |
| "eval_runtime": 25.2334, | |
| "eval_samples_per_second": 5.826, | |
| "eval_steps_per_second": 1.466, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 0.3463636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.321962215109906e-07, | |
| "loss": 3.015147590637207, | |
| "step": 3810 | |
| }, | |
| { | |
| "epoch": 0.3472727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.309306002466787e-07, | |
| "loss": 3.0093734741210936, | |
| "step": 3820 | |
| }, | |
| { | |
| "epoch": 0.3481818181818182, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.296630953525158e-07, | |
| "loss": 2.9403078079223635, | |
| "step": 3830 | |
| }, | |
| { | |
| "epoch": 0.3490909090909091, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.283937171671497e-07, | |
| "loss": 2.6926576614379885, | |
| "step": 3840 | |
| }, | |
| { | |
| "epoch": 0.35, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.271224760445079e-07, | |
| "loss": 2.754840850830078, | |
| "step": 3850 | |
| }, | |
| { | |
| "epoch": 0.3509090909090909, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.258493823537124e-07, | |
| "loss": 2.694375419616699, | |
| "step": 3860 | |
| }, | |
| { | |
| "epoch": 0.3518181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.245744464789973e-07, | |
| "loss": 2.945112419128418, | |
| "step": 3870 | |
| }, | |
| { | |
| "epoch": 0.3527272727272727, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.232976788196221e-07, | |
| "loss": 2.8494308471679686, | |
| "step": 3880 | |
| }, | |
| { | |
| "epoch": 0.35363636363636364, | |
| "grad_norm": 1.6991103928231789e+19, | |
| "learning_rate": 7.220190897897877e-07, | |
| "loss": 2.9483663558959963, | |
| "step": 3890 | |
| }, | |
| { | |
| "epoch": 0.35454545454545455, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.207386898185515e-07, | |
| "loss": 2.6762250900268554, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 0.35454545454545455, | |
| "eval_loss": 2.7912540435791016, | |
| "eval_runtime": 25.2005, | |
| "eval_samples_per_second": 5.833, | |
| "eval_steps_per_second": 1.468, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 0.35545454545454547, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.194564893497419e-07, | |
| "loss": 2.797460174560547, | |
| "step": 3910 | |
| }, | |
| { | |
| "epoch": 0.3563636363636364, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.181724988418737e-07, | |
| "loss": 3.264780044555664, | |
| "step": 3920 | |
| }, | |
| { | |
| "epoch": 0.3572727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.168867287680627e-07, | |
| "loss": 2.6318002700805665, | |
| "step": 3930 | |
| }, | |
| { | |
| "epoch": 0.35818181818181816, | |
| "grad_norm": 3.8170838084406477e+18, | |
| "learning_rate": 7.155991896159392e-07, | |
| "loss": 2.9560144424438475, | |
| "step": 3940 | |
| }, | |
| { | |
| "epoch": 0.35909090909090907, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.143098918875642e-07, | |
| "loss": 2.273042678833008, | |
| "step": 3950 | |
| }, | |
| { | |
| "epoch": 0.36, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.130188460993426e-07, | |
| "loss": 2.5968145370483398, | |
| "step": 3960 | |
| }, | |
| { | |
| "epoch": 0.3609090909090909, | |
| "grad_norm": 1.2694314741464564e+19, | |
| "learning_rate": 7.117260627819376e-07, | |
| "loss": 2.8214059829711915, | |
| "step": 3970 | |
| }, | |
| { | |
| "epoch": 0.3618181818181818, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.104315524801848e-07, | |
| "loss": 2.538612174987793, | |
| "step": 3980 | |
| }, | |
| { | |
| "epoch": 0.36272727272727273, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.091353257530068e-07, | |
| "loss": 2.7980751037597655, | |
| "step": 3990 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "grad_norm": Infinity, | |
| "learning_rate": 7.078373931733258e-07, | |
| "loss": 2.8797430038452148, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "eval_loss": 2.791391372680664, | |
| "eval_runtime": 25.1579, | |
| "eval_samples_per_second": 5.843, | |
| "eval_steps_per_second": 1.471, | |
| "step": 4000 | |
| } | |
| ], | |
| "logging_steps": 10, | |
| "max_steps": 11000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 4000, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 4529923817472000.0, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |