Download checkpoint-2000/trainer_state.json from Nihilux/SpringHunter: direct link, hf CLI and curl.
- Browser
- Download file 39.2 kB
-
https://huggingface.co/Nihilux/SpringHunter/resolve/main/checkpoint-2000/trainer_state.json
- Command line
-
hf download hf://Nihilux/SpringHunter/checkpoint-2000/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/Nihilux/SpringHunter/resolve/main/checkpoint-2000/trainer_state.json
39.2 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.7717258926095187, | |
| "eval_steps": 150, | |
| "global_step": 2000, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.003858629463047594, | |
| "grad_norm": 0.9612510204315186, | |
| "learning_rate": 1.8e-05, | |
| "loss": 1.0067429542541504, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.007717258926095188, | |
| "grad_norm": 0.6660895347595215, | |
| "learning_rate": 3.8e-05, | |
| "loss": 0.8466403007507324, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.011575888389142782, | |
| "grad_norm": 0.5681023001670837, | |
| "learning_rate": 5.8e-05, | |
| "loss": 0.7557645797729492, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.015434517852190376, | |
| "grad_norm": 0.7179134488105774, | |
| "learning_rate": 7.800000000000001e-05, | |
| "loss": 0.7409078121185303, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.019293147315237968, | |
| "grad_norm": 0.5565946102142334, | |
| "learning_rate": 9.8e-05, | |
| "loss": 0.7274651050567627, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.023151776778285563, | |
| "grad_norm": 0.5450749397277832, | |
| "learning_rate": 0.000118, | |
| "loss": 0.741389274597168, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.027010406241333156, | |
| "grad_norm": 0.4871410131454468, | |
| "learning_rate": 0.000138, | |
| "loss": 0.725303316116333, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.03086903570438075, | |
| "grad_norm": 0.5452415347099304, | |
| "learning_rate": 0.00015800000000000002, | |
| "loss": 0.724397611618042, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.03472766516742835, | |
| "grad_norm": 0.438076376914978, | |
| "learning_rate": 0.00017800000000000002, | |
| "loss": 0.7216084480285645, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.038586294630475935, | |
| "grad_norm": 0.3792989253997803, | |
| "learning_rate": 0.00019800000000000002, | |
| "loss": 0.7241204738616943, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.04244492409352353, | |
| "grad_norm": 0.31958940625190735, | |
| "learning_rate": 0.0001999935634368633, | |
| "loss": 0.7156620502471924, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.04630355355657113, | |
| "grad_norm": 0.3532995283603668, | |
| "learning_rate": 0.00019997131465276176, | |
| "loss": 0.716402530670166, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.05016218301961872, | |
| "grad_norm": 0.3720763921737671, | |
| "learning_rate": 0.0001999331777190454, | |
| "loss": 0.725121021270752, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.05402081248266631, | |
| "grad_norm": 0.327168345451355, | |
| "learning_rate": 0.0001998791586967059, | |
| "loss": 0.711461877822876, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.05787944194571391, | |
| "grad_norm": 0.31293419003486633, | |
| "learning_rate": 0.00019980926617082901, | |
| "loss": 0.7175386905670166, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.05787944194571391, | |
| "eval_loss": 0.7297696471214294, | |
| "eval_runtime": 244.1933, | |
| "eval_samples_per_second": 3.419, | |
| "eval_steps_per_second": 1.712, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.0617380714087615, | |
| "grad_norm": 0.3199326694011688, | |
| "learning_rate": 0.0001997235112492302, | |
| "loss": 0.7084239006042481, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.06559670087180909, | |
| "grad_norm": 0.3334254026412964, | |
| "learning_rate": 0.00019962190756068907, | |
| "loss": 0.7120396614074707, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.0694553303348567, | |
| "grad_norm": 0.32970502972602844, | |
| "learning_rate": 0.00019950447125278376, | |
| "loss": 0.7141448974609375, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.07331395979790428, | |
| "grad_norm": 0.3339720368385315, | |
| "learning_rate": 0.00019937122098932428, | |
| "loss": 0.7231975555419922, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.07717258926095187, | |
| "grad_norm": 0.302263081073761, | |
| "learning_rate": 0.00019922217794738657, | |
| "loss": 0.7377682685852051, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.08103121872399947, | |
| "grad_norm": 0.3148849904537201, | |
| "learning_rate": 0.00019905736581394689, | |
| "loss": 0.7064791202545166, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.08488984818704706, | |
| "grad_norm": 0.3036375641822815, | |
| "learning_rate": 0.00019887681078211707, | |
| "loss": 0.7082621097564697, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.08874847765009465, | |
| "grad_norm": 0.353236585855484, | |
| "learning_rate": 0.00019868054154698202, | |
| "loss": 0.7202902793884277, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.09260710711314225, | |
| "grad_norm": 0.2983817458152771, | |
| "learning_rate": 0.0001984685893010392, | |
| "loss": 0.6949288845062256, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.09646573657618984, | |
| "grad_norm": 0.3325659930706024, | |
| "learning_rate": 0.00019824098772924114, | |
| "loss": 0.7241635799407959, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.10032436603923744, | |
| "grad_norm": 0.29093560576438904, | |
| "learning_rate": 0.00019799777300364218, | |
| "loss": 0.7079925060272216, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.10418299550228503, | |
| "grad_norm": 0.3151165246963501, | |
| "learning_rate": 0.0001977389837776497, | |
| "loss": 0.7299640655517579, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.10804162496533262, | |
| "grad_norm": 0.30055317282676697, | |
| "learning_rate": 0.000197464661179881, | |
| "loss": 0.7072536468505859, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.11190025442838022, | |
| "grad_norm": 0.28629517555236816, | |
| "learning_rate": 0.00019717484880762685, | |
| "loss": 0.7206380844116211, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.1196175133544754, | |
| "grad_norm": 0.31153738498687744, | |
| "learning_rate": 0.0001965489414302289, | |
| "loss": 0.722527265548706, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.123476142817523, | |
| "grad_norm": 0.2931964695453644, | |
| "learning_rate": 0.00019621294589872003, | |
| "loss": 0.7037133693695068, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.1273347722805706, | |
| "grad_norm": 0.33466067910194397, | |
| "learning_rate": 0.0001958616595241865, | |
| "loss": 0.727421760559082, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.13119340174361818, | |
| "grad_norm": 0.28575319051742554, | |
| "learning_rate": 0.00019549513813554785, | |
| "loss": 0.6994576930999756, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.13505203120666578, | |
| "grad_norm": 0.31457364559173584, | |
| "learning_rate": 0.0001951134399829799, | |
| "loss": 0.734303331375122, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.1389106606697134, | |
| "grad_norm": 0.30264604091644287, | |
| "learning_rate": 0.00019471662572865736, | |
| "loss": 0.7409284591674805, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.14276929013276096, | |
| "grad_norm": 0.28342851996421814, | |
| "learning_rate": 0.00019430475843711293, | |
| "loss": 0.7335753440856934, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.14662791959580856, | |
| "grad_norm": 0.30577903985977173, | |
| "learning_rate": 0.00019387790356521463, | |
| "loss": 0.7155701637268066, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.15048654905885617, | |
| "grad_norm": 0.28058379888534546, | |
| "learning_rate": 0.000193436128951763, | |
| "loss": 0.6958878993988037, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.15434517852190374, | |
| "grad_norm": 0.30901074409484863, | |
| "learning_rate": 0.0001929795048067095, | |
| "loss": 0.7224118709564209, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.15820380798495134, | |
| "grad_norm": 0.28316858410835266, | |
| "learning_rate": 0.0001925081036999984, | |
| "loss": 0.7156408786773681, | |
| "step": 410 | |
| }, | |
| { | |
| "epoch": 0.16206243744799895, | |
| "grad_norm": 0.29417017102241516, | |
| "learning_rate": 0.00019202200055003346, | |
| "loss": 0.7048601150512696, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.16592106691104652, | |
| "grad_norm": 0.29546213150024414, | |
| "learning_rate": 0.00019152127261177126, | |
| "loss": 0.7047354698181152, | |
| "step": 430 | |
| }, | |
| { | |
| "epoch": 0.16977969637409412, | |
| "grad_norm": 0.2986455261707306, | |
| "learning_rate": 0.0001910059994644434, | |
| "loss": 0.6936575889587402, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 0.17363832583714173, | |
| "grad_norm": 0.35280054807662964, | |
| "learning_rate": 0.0001904762629989091, | |
| "loss": 0.6981217384338378, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.17363832583714173, | |
| "eval_loss": 0.7733834385871887, | |
| "eval_runtime": 241.1126, | |
| "eval_samples_per_second": 3.463, | |
| "eval_steps_per_second": 1.734, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.1774969553001893, | |
| "grad_norm": 0.28504979610443115, | |
| "learning_rate": 0.00018993214740464063, | |
| "loss": 0.7187217235565185, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 0.1813555847632369, | |
| "grad_norm": 0.294179230928421, | |
| "learning_rate": 0.00018937373915634323, | |
| "loss": 0.723856258392334, | |
| "step": 470 | |
| }, | |
| { | |
| "epoch": 0.1852142142262845, | |
| "grad_norm": 0.3039489984512329, | |
| "learning_rate": 0.00018880112700021205, | |
| "loss": 0.7158075332641601, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 0.18907284368933208, | |
| "grad_norm": 0.3352907598018646, | |
| "learning_rate": 0.0001882144019398278, | |
| "loss": 0.7123313903808594, | |
| "step": 490 | |
| }, | |
| { | |
| "epoch": 0.19293147315237968, | |
| "grad_norm": 0.31155890226364136, | |
| "learning_rate": 0.00018761365722169403, | |
| "loss": 0.7325988292694092, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.1967901026154273, | |
| "grad_norm": 0.3036295175552368, | |
| "learning_rate": 0.00018699898832041757, | |
| "loss": 0.7070968627929688, | |
| "step": 510 | |
| }, | |
| { | |
| "epoch": 0.2006487320784749, | |
| "grad_norm": 0.2792280316352844, | |
| "learning_rate": 0.00018637049292353513, | |
| "loss": 0.7023006916046143, | |
| "step": 520 | |
| }, | |
| { | |
| "epoch": 0.20450736154152246, | |
| "grad_norm": 0.30598390102386475, | |
| "learning_rate": 0.00018572827091598793, | |
| "loss": 0.708428144454956, | |
| "step": 530 | |
| }, | |
| { | |
| "epoch": 0.20836599100457007, | |
| "grad_norm": 0.28768792748451233, | |
| "learning_rate": 0.00018507242436424765, | |
| "loss": 0.708616304397583, | |
| "step": 540 | |
| }, | |
| { | |
| "epoch": 0.21222462046761767, | |
| "grad_norm": 0.32237133383750916, | |
| "learning_rate": 0.00018440305750009483, | |
| "loss": 0.7210727214813233, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 0.21608324993066524, | |
| "grad_norm": 0.2892608940601349, | |
| "learning_rate": 0.000183720276704054, | |
| "loss": 0.6931821823120117, | |
| "step": 560 | |
| }, | |
| { | |
| "epoch": 0.21994187939371285, | |
| "grad_norm": 0.3152562081813812, | |
| "learning_rate": 0.00018302419048848667, | |
| "loss": 0.7213119506835938, | |
| "step": 570 | |
| }, | |
| { | |
| "epoch": 0.22380050885676045, | |
| "grad_norm": 0.31499791145324707, | |
| "learning_rate": 0.00018231490948034592, | |
| "loss": 0.6939045906066894, | |
| "step": 580 | |
| }, | |
| { | |
| "epoch": 0.22765913831980802, | |
| "grad_norm": 0.2861924469470978, | |
| "learning_rate": 0.00018159254640359487, | |
| "loss": 0.7252971649169921, | |
| "step": 590 | |
| }, | |
| { | |
| "epoch": 0.23537639724590323, | |
| "grad_norm": 0.2836938500404358, | |
| "learning_rate": 0.00018010903531734363, | |
| "loss": 0.6902801513671875, | |
| "step": 610 | |
| }, | |
| { | |
| "epoch": 0.2392350267089508, | |
| "grad_norm": 0.3197932243347168, | |
| "learning_rate": 0.00017934812307793583, | |
| "loss": 0.6933451175689698, | |
| "step": 620 | |
| }, | |
| { | |
| "epoch": 0.2430936561719984, | |
| "grad_norm": 0.29024383425712585, | |
| "learning_rate": 0.00017857460027263225, | |
| "loss": 0.6954158306121826, | |
| "step": 630 | |
| }, | |
| { | |
| "epoch": 0.246952285635046, | |
| "grad_norm": 0.29053837060928345, | |
| "learning_rate": 0.00017778858983515743, | |
| "loss": 0.7062996864318848, | |
| "step": 640 | |
| }, | |
| { | |
| "epoch": 0.2508109150980936, | |
| "grad_norm": 0.2862823009490967, | |
| "learning_rate": 0.00017699021668385895, | |
| "loss": 0.726755428314209, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 0.2546695445611412, | |
| "grad_norm": 0.31322675943374634, | |
| "learning_rate": 0.00017617960770185444, | |
| "loss": 0.7403687953948974, | |
| "step": 660 | |
| }, | |
| { | |
| "epoch": 0.25852817402418876, | |
| "grad_norm": 0.2916722595691681, | |
| "learning_rate": 0.00017535689171686644, | |
| "loss": 0.6931378364562988, | |
| "step": 670 | |
| }, | |
| { | |
| "epoch": 0.26238680348723636, | |
| "grad_norm": 0.3099428117275238, | |
| "learning_rate": 0.00017452219948074814, | |
| "loss": 0.7085556983947754, | |
| "step": 680 | |
| }, | |
| { | |
| "epoch": 0.26624543295028397, | |
| "grad_norm": 0.30417969822883606, | |
| "learning_rate": 0.0001736756636487035, | |
| "loss": 0.7230380058288575, | |
| "step": 690 | |
| }, | |
| { | |
| "epoch": 0.27010406241333157, | |
| "grad_norm": 0.28197744488716125, | |
| "learning_rate": 0.0001728174187582045, | |
| "loss": 0.722746467590332, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.27396269187637917, | |
| "grad_norm": 0.30067917704582214, | |
| "learning_rate": 0.00017194760120760986, | |
| "loss": 0.7116784572601318, | |
| "step": 710 | |
| }, | |
| { | |
| "epoch": 0.2778213213394268, | |
| "grad_norm": 0.2828962802886963, | |
| "learning_rate": 0.00017106634923448724, | |
| "loss": 0.6991414546966552, | |
| "step": 720 | |
| }, | |
| { | |
| "epoch": 0.2816799508024743, | |
| "grad_norm": 0.303771436214447, | |
| "learning_rate": 0.00017017380289364388, | |
| "loss": 0.700380277633667, | |
| "step": 730 | |
| }, | |
| { | |
| "epoch": 0.2855385802655219, | |
| "grad_norm": 0.3401305377483368, | |
| "learning_rate": 0.00016927010403486786, | |
| "loss": 0.717545461654663, | |
| "step": 740 | |
| }, | |
| { | |
| "epoch": 0.2893972097285695, | |
| "grad_norm": 0.29057246446609497, | |
| "learning_rate": 0.00016835539628038445, | |
| "loss": 0.7112515449523926, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 0.2893972097285695, | |
| "eval_loss": 0.7159802317619324, | |
| "eval_runtime": 247.114, | |
| "eval_samples_per_second": 3.379, | |
| "eval_steps_per_second": 1.692, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 0.29325583919161713, | |
| "grad_norm": 0.2816619277000427, | |
| "learning_rate": 0.0001674298250020307, | |
| "loss": 0.7210293769836426, | |
| "step": 760 | |
| }, | |
| { | |
| "epoch": 0.29711446865466473, | |
| "grad_norm": 0.35117030143737793, | |
| "learning_rate": 0.00016649353729815172, | |
| "loss": 0.7004368305206299, | |
| "step": 770 | |
| }, | |
| { | |
| "epoch": 0.30097309811771233, | |
| "grad_norm": 0.3049272298812866, | |
| "learning_rate": 0.00016554668197022295, | |
| "loss": 0.693110990524292, | |
| "step": 780 | |
| }, | |
| { | |
| "epoch": 0.3048317275807599, | |
| "grad_norm": 0.33454808592796326, | |
| "learning_rate": 0.0001645894094992015, | |
| "loss": 0.7131325244903565, | |
| "step": 790 | |
| }, | |
| { | |
| "epoch": 0.3086903570438075, | |
| "grad_norm": 0.2931094765663147, | |
| "learning_rate": 0.00016362187202161076, | |
| "loss": 0.7059287071228028, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.3125489865068551, | |
| "grad_norm": 0.31233930587768555, | |
| "learning_rate": 0.00016264422330536154, | |
| "loss": 0.7155674934387207, | |
| "step": 810 | |
| }, | |
| { | |
| "epoch": 0.3164076159699027, | |
| "grad_norm": 0.30221912264823914, | |
| "learning_rate": 0.00016165661872531443, | |
| "loss": 0.7006759166717529, | |
| "step": 820 | |
| }, | |
| { | |
| "epoch": 0.3202662454329503, | |
| "grad_norm": 0.28081852197647095, | |
| "learning_rate": 0.00016065921523858635, | |
| "loss": 0.7008425712585449, | |
| "step": 830 | |
| }, | |
| { | |
| "epoch": 0.3241248748959979, | |
| "grad_norm": 0.31421926617622375, | |
| "learning_rate": 0.00015965217135960607, | |
| "loss": 0.7249965190887451, | |
| "step": 840 | |
| }, | |
| { | |
| "epoch": 0.3279835043590455, | |
| "grad_norm": 0.2977774143218994, | |
| "learning_rate": 0.0001586356471349215, | |
| "loss": 0.6932929039001465, | |
| "step": 850 | |
| }, | |
| { | |
| "epoch": 0.33184213382209304, | |
| "grad_norm": 0.28536227345466614, | |
| "learning_rate": 0.0001576098041177646, | |
| "loss": 0.7123911857604981, | |
| "step": 860 | |
| }, | |
| { | |
| "epoch": 0.33570076328514065, | |
| "grad_norm": 0.2825419008731842, | |
| "learning_rate": 0.00015657480534237561, | |
| "loss": 0.6955167293548584, | |
| "step": 870 | |
| }, | |
| { | |
| "epoch": 0.33955939274818825, | |
| "grad_norm": 0.2923476994037628, | |
| "learning_rate": 0.00015553081529809281, | |
| "loss": 0.6951467990875244, | |
| "step": 880 | |
| }, | |
| { | |
| "epoch": 0.34341802221123585, | |
| "grad_norm": 0.290323406457901, | |
| "learning_rate": 0.00015447799990321066, | |
| "loss": 0.6957611083984375, | |
| "step": 890 | |
| }, | |
| { | |
| "epoch": 0.34727665167428345, | |
| "grad_norm": 0.29604029655456543, | |
| "learning_rate": 0.00015341652647861084, | |
| "loss": 0.7186079025268555, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.34727665167428345, | |
| "eval_loss": 0.7107371687889099, | |
| "eval_runtime": 261.1674, | |
| "eval_samples_per_second": 3.197, | |
| "eval_steps_per_second": 1.601, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.35113528113733106, | |
| "grad_norm": 0.27018797397613525, | |
| "learning_rate": 0.00015234656372117042, | |
| "loss": 0.7154855728149414, | |
| "step": 910 | |
| }, | |
| { | |
| "epoch": 0.3549939106003786, | |
| "grad_norm": 0.26977527141571045, | |
| "learning_rate": 0.00015126828167695146, | |
| "loss": 0.699330997467041, | |
| "step": 920 | |
| }, | |
| { | |
| "epoch": 0.3588525400634262, | |
| "grad_norm": 0.30887559056282043, | |
| "learning_rate": 0.00015018185171417604, | |
| "loss": 0.6892439842224121, | |
| "step": 930 | |
| }, | |
| { | |
| "epoch": 0.3627111695264738, | |
| "grad_norm": 0.31451812386512756, | |
| "learning_rate": 0.00014908744649599104, | |
| "loss": 0.7145347595214844, | |
| "step": 940 | |
| }, | |
| { | |
| "epoch": 0.3665697989895214, | |
| "grad_norm": 0.2898563742637634, | |
| "learning_rate": 0.00014798523995302758, | |
| "loss": 0.7055789947509765, | |
| "step": 950 | |
| }, | |
| { | |
| "epoch": 0.370428428452569, | |
| "grad_norm": 0.26642370223999023, | |
| "learning_rate": 0.00014687540725575845, | |
| "loss": 0.6984900951385498, | |
| "step": 960 | |
| }, | |
| { | |
| "epoch": 0.3742870579156166, | |
| "grad_norm": 0.32259368896484375, | |
| "learning_rate": 0.0001457581247866591, | |
| "loss": 0.6736190795898438, | |
| "step": 970 | |
| }, | |
| { | |
| "epoch": 0.37814568737866416, | |
| "grad_norm": 0.2674982249736786, | |
| "learning_rate": 0.00014463357011217532, | |
| "loss": 0.6823287010192871, | |
| "step": 980 | |
| }, | |
| { | |
| "epoch": 0.38200431684171177, | |
| "grad_norm": 0.295978307723999, | |
| "learning_rate": 0.0001435019219545034, | |
| "loss": 0.7077735900878906, | |
| "step": 990 | |
| }, | |
| { | |
| "epoch": 0.38586294630475937, | |
| "grad_norm": 0.3013753294944763, | |
| "learning_rate": 0.0001423633601631862, | |
| "loss": 0.7043970108032227, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.38972157576780697, | |
| "grad_norm": 0.26629018783569336, | |
| "learning_rate": 0.00014121806568653025, | |
| "loss": 0.6968683719635009, | |
| "step": 1010 | |
| }, | |
| { | |
| "epoch": 0.3935802052308546, | |
| "grad_norm": 0.2660974860191345, | |
| "learning_rate": 0.00014006622054284806, | |
| "loss": 0.6901365756988526, | |
| "step": 1020 | |
| }, | |
| { | |
| "epoch": 0.3974388346939022, | |
| "grad_norm": 0.2733144760131836, | |
| "learning_rate": 0.0001389080077915307, | |
| "loss": 0.6891340255737305, | |
| "step": 1030 | |
| }, | |
| { | |
| "epoch": 0.4012974641569498, | |
| "grad_norm": 0.30119362473487854, | |
| "learning_rate": 0.00013774361150395442, | |
| "loss": 0.687501335144043, | |
| "step": 1040 | |
| }, | |
| { | |
| "epoch": 0.4051560936199973, | |
| "grad_norm": 0.29165130853652954, | |
| "learning_rate": 0.00013657321673422691, | |
| "loss": 0.7198416233062744, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 0.4051560936199973, | |
| "eval_loss": 0.7047804594039917, | |
| "eval_runtime": 261.4782, | |
| "eval_samples_per_second": 3.193, | |
| "eval_steps_per_second": 1.599, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 0.40901472308304493, | |
| "grad_norm": 0.2916712760925293, | |
| "learning_rate": 0.00013539700948977717, | |
| "loss": 0.6978848934173584, | |
| "step": 1060 | |
| }, | |
| { | |
| "epoch": 0.41287335254609253, | |
| "grad_norm": 0.29535216093063354, | |
| "learning_rate": 0.0001342151767017938, | |
| "loss": 0.6890017986297607, | |
| "step": 1070 | |
| }, | |
| { | |
| "epoch": 0.41673198200914013, | |
| "grad_norm": 0.28752076625823975, | |
| "learning_rate": 0.00013302790619551674, | |
| "loss": 0.717583703994751, | |
| "step": 1080 | |
| }, | |
| { | |
| "epoch": 0.42059061147218774, | |
| "grad_norm": 0.32082512974739075, | |
| "learning_rate": 0.00013183538666038648, | |
| "loss": 0.7101277351379395, | |
| "step": 1090 | |
| }, | |
| { | |
| "epoch": 0.42444924093523534, | |
| "grad_norm": 0.27900826930999756, | |
| "learning_rate": 0.00013063780762005654, | |
| "loss": 0.6807409286499023, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.4283078703982829, | |
| "grad_norm": 0.3178718090057373, | |
| "learning_rate": 0.0001294353594022728, | |
| "loss": 0.686630392074585, | |
| "step": 1110 | |
| }, | |
| { | |
| "epoch": 0.4321664998613305, | |
| "grad_norm": 0.2647388279438019, | |
| "learning_rate": 0.00012822823310862524, | |
| "loss": 0.670403003692627, | |
| "step": 1120 | |
| }, | |
| { | |
| "epoch": 0.4360251293243781, | |
| "grad_norm": 0.28969231247901917, | |
| "learning_rate": 0.00012701662058417688, | |
| "loss": 0.6713655948638916, | |
| "step": 1130 | |
| }, | |
| { | |
| "epoch": 0.4398837587874257, | |
| "grad_norm": 0.28609490394592285, | |
| "learning_rate": 0.00012580071438697427, | |
| "loss": 0.6948023319244385, | |
| "step": 1140 | |
| }, | |
| { | |
| "epoch": 0.4437423882504733, | |
| "grad_norm": 0.27543318271636963, | |
| "learning_rate": 0.0001245807077574449, | |
| "loss": 0.7000330924987793, | |
| "step": 1150 | |
| }, | |
| { | |
| "epoch": 0.4476010177135209, | |
| "grad_norm": 0.28665491938591003, | |
| "learning_rate": 0.00012335679458768607, | |
| "loss": 0.6907845497131347, | |
| "step": 1160 | |
| }, | |
| { | |
| "epoch": 0.45145964717656845, | |
| "grad_norm": 0.26809993386268616, | |
| "learning_rate": 0.00012212916939064998, | |
| "loss": 0.6751095294952393, | |
| "step": 1170 | |
| }, | |
| { | |
| "epoch": 0.45531827663961605, | |
| "grad_norm": 0.28608420491218567, | |
| "learning_rate": 0.00012089802726923061, | |
| "loss": 0.6977941513061523, | |
| "step": 1180 | |
| }, | |
| { | |
| "epoch": 0.45917690610266365, | |
| "grad_norm": 0.2964702844619751, | |
| "learning_rate": 0.00011966356388525646, | |
| "loss": 0.685225248336792, | |
| "step": 1190 | |
| }, | |
| { | |
| "epoch": 0.46689416502875886, | |
| "grad_norm": 0.3035467863082886, | |
| "learning_rate": 0.00011718545858497087, | |
| "loss": 0.6892534732818604, | |
| "step": 1210 | |
| }, | |
| { | |
| "epoch": 0.47075279449180646, | |
| "grad_norm": 0.2953267991542816, | |
| "learning_rate": 0.00011594221050671095, | |
| "loss": 0.6970641136169433, | |
| "step": 1220 | |
| }, | |
| { | |
| "epoch": 0.47461142395485406, | |
| "grad_norm": 0.2962872087955475, | |
| "learning_rate": 0.00011469642877940779, | |
| "loss": 0.6811253547668457, | |
| "step": 1230 | |
| }, | |
| { | |
| "epoch": 0.4784700534179016, | |
| "grad_norm": 0.2830030024051666, | |
| "learning_rate": 0.00011344831139151972, | |
| "loss": 0.6688879489898681, | |
| "step": 1240 | |
| }, | |
| { | |
| "epoch": 0.4823286828809492, | |
| "grad_norm": 0.29539620876312256, | |
| "learning_rate": 0.00011219805670270496, | |
| "loss": 0.700579833984375, | |
| "step": 1250 | |
| }, | |
| { | |
| "epoch": 0.4861873123439968, | |
| "grad_norm": 0.29297176003456116, | |
| "learning_rate": 0.00011094586341229656, | |
| "loss": 0.6867257118225097, | |
| "step": 1260 | |
| }, | |
| { | |
| "epoch": 0.4900459418070444, | |
| "grad_norm": 0.29863327741622925, | |
| "learning_rate": 0.00010969193052772396, | |
| "loss": 0.71397385597229, | |
| "step": 1270 | |
| }, | |
| { | |
| "epoch": 0.493904571270092, | |
| "grad_norm": 0.2799651622772217, | |
| "learning_rate": 0.00010843645733288519, | |
| "loss": 0.7007513046264648, | |
| "step": 1280 | |
| }, | |
| { | |
| "epoch": 0.4977632007331396, | |
| "grad_norm": 0.2826097905635834, | |
| "learning_rate": 0.00010717964335647535, | |
| "loss": 0.6815872669219971, | |
| "step": 1290 | |
| }, | |
| { | |
| "epoch": 0.5016218301961872, | |
| "grad_norm": 0.27504852414131165, | |
| "learning_rate": 0.00010592168834027598, | |
| "loss": 0.6807018280029297, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.5054804596592348, | |
| "grad_norm": 0.28094950318336487, | |
| "learning_rate": 0.00010466279220741078, | |
| "loss": 0.6823868751525879, | |
| "step": 1310 | |
| }, | |
| { | |
| "epoch": 0.5093390891222824, | |
| "grad_norm": 0.2797650694847107, | |
| "learning_rate": 0.00010340315503057243, | |
| "loss": 0.674619722366333, | |
| "step": 1320 | |
| }, | |
| { | |
| "epoch": 0.5131977185853299, | |
| "grad_norm": 0.28894102573394775, | |
| "learning_rate": 0.00010214297700022549, | |
| "loss": 0.7022358417510987, | |
| "step": 1330 | |
| }, | |
| { | |
| "epoch": 0.5170563480483775, | |
| "grad_norm": 0.29386165738105774, | |
| "learning_rate": 0.00010088245839279082, | |
| "loss": 0.67178316116333, | |
| "step": 1340 | |
| }, | |
| { | |
| "epoch": 0.5209149775114251, | |
| "grad_norm": 0.28537702560424805, | |
| "learning_rate": 9.96217995388162e-05, | |
| "loss": 0.6890518188476562, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 0.5209149775114251, | |
| "eval_loss": 0.6932345032691956, | |
| "eval_runtime": 269.3818, | |
| "eval_samples_per_second": 3.1, | |
| "eval_steps_per_second": 1.552, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 0.5247736069744727, | |
| "grad_norm": 0.3049728274345398, | |
| "learning_rate": 9.83612007911384e-05, | |
| "loss": 0.6744989871978759, | |
| "step": 1360 | |
| }, | |
| { | |
| "epoch": 0.5286322364375203, | |
| "grad_norm": 0.33253172039985657, | |
| "learning_rate": 9.710086249304164e-05, | |
| "loss": 0.6773325443267822, | |
| "step": 1370 | |
| }, | |
| { | |
| "epoch": 0.5324908659005679, | |
| "grad_norm": 0.27811330556869507, | |
| "learning_rate": 9.584098494641772e-05, | |
| "loss": 0.6876926898956299, | |
| "step": 1380 | |
| }, | |
| { | |
| "epoch": 0.5363494953636155, | |
| "grad_norm": 0.29328352212905884, | |
| "learning_rate": 9.458176837993246e-05, | |
| "loss": 0.689525842666626, | |
| "step": 1390 | |
| }, | |
| { | |
| "epoch": 0.5402081248266631, | |
| "grad_norm": 0.27268341183662415, | |
| "learning_rate": 9.332341291720408e-05, | |
| "loss": 0.6747959136962891, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.5440667542897107, | |
| "grad_norm": 0.28394851088523865, | |
| "learning_rate": 9.206611854499805e-05, | |
| "loss": 0.6711764335632324, | |
| "step": 1410 | |
| }, | |
| { | |
| "epoch": 0.5479253837527583, | |
| "grad_norm": 0.28641200065612793, | |
| "learning_rate": 9.081008508144388e-05, | |
| "loss": 0.6956794261932373, | |
| "step": 1420 | |
| }, | |
| { | |
| "epoch": 0.551784013215806, | |
| "grad_norm": 0.2705199718475342, | |
| "learning_rate": 8.955551214427856e-05, | |
| "loss": 0.7046959400177002, | |
| "step": 1430 | |
| }, | |
| { | |
| "epoch": 0.5556426426788535, | |
| "grad_norm": 0.26797372102737427, | |
| "learning_rate": 8.830259911912173e-05, | |
| "loss": 0.6727779865264892, | |
| "step": 1440 | |
| }, | |
| { | |
| "epoch": 0.5595012721419012, | |
| "grad_norm": 0.2805560529232025, | |
| "learning_rate": 8.705154512778821e-05, | |
| "loss": 0.6727550506591797, | |
| "step": 1450 | |
| }, | |
| { | |
| "epoch": 0.5633599016049486, | |
| "grad_norm": 0.28046268224716187, | |
| "learning_rate": 8.580254899664195e-05, | |
| "loss": 0.6817981243133545, | |
| "step": 1460 | |
| }, | |
| { | |
| "epoch": 0.5672185310679962, | |
| "grad_norm": 0.27203667163848877, | |
| "learning_rate": 8.455580922499716e-05, | |
| "loss": 0.6813424110412598, | |
| "step": 1470 | |
| }, | |
| { | |
| "epoch": 0.5710771605310438, | |
| "grad_norm": 0.29054710268974304, | |
| "learning_rate": 8.331152395357141e-05, | |
| "loss": 0.6807193279266357, | |
| "step": 1480 | |
| }, | |
| { | |
| "epoch": 0.5749357899940915, | |
| "grad_norm": 0.32792478799819946, | |
| "learning_rate": 8.206989093299572e-05, | |
| "loss": 0.6816314220428467, | |
| "step": 1490 | |
| }, | |
| { | |
| "epoch": 0.578794419457139, | |
| "grad_norm": 0.2703556418418884, | |
| "learning_rate": 8.083110749238659e-05, | |
| "loss": 0.6654800415039063, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.578794419457139, | |
| "eval_loss": 0.6862806081771851, | |
| "eval_runtime": 245.6426, | |
| "eval_samples_per_second": 3.399, | |
| "eval_steps_per_second": 1.702, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.5826530489201867, | |
| "grad_norm": 0.27336329221725464, | |
| "learning_rate": 7.959537050798512e-05, | |
| "loss": 0.6732516288757324, | |
| "step": 1510 | |
| }, | |
| { | |
| "epoch": 0.5865116783832343, | |
| "grad_norm": 0.27847999334335327, | |
| "learning_rate": 7.836287637186801e-05, | |
| "loss": 0.68833327293396, | |
| "step": 1520 | |
| }, | |
| { | |
| "epoch": 0.5903703078462819, | |
| "grad_norm": 0.25630271434783936, | |
| "learning_rate": 7.713382096073545e-05, | |
| "loss": 0.6711580276489257, | |
| "step": 1530 | |
| }, | |
| { | |
| "epoch": 0.5942289373093295, | |
| "grad_norm": 0.2752869427204132, | |
| "learning_rate": 7.59083996047812e-05, | |
| "loss": 0.6594626426696777, | |
| "step": 1540 | |
| }, | |
| { | |
| "epoch": 0.5980875667723771, | |
| "grad_norm": 0.27679699659347534, | |
| "learning_rate": 7.468680705664914e-05, | |
| "loss": 0.6733936309814453, | |
| "step": 1550 | |
| }, | |
| { | |
| "epoch": 0.6019461962354247, | |
| "grad_norm": 0.2734837830066681, | |
| "learning_rate": 7.346923746048202e-05, | |
| "loss": 0.6738894939422607, | |
| "step": 1560 | |
| }, | |
| { | |
| "epoch": 0.6058048256984723, | |
| "grad_norm": 0.27832263708114624, | |
| "learning_rate": 7.225588432106633e-05, | |
| "loss": 0.6458727359771729, | |
| "step": 1570 | |
| }, | |
| { | |
| "epoch": 0.6096634551615198, | |
| "grad_norm": 0.28797298669815063, | |
| "learning_rate": 7.104694047307963e-05, | |
| "loss": 0.6783483982086181, | |
| "step": 1580 | |
| }, | |
| { | |
| "epoch": 0.6135220846245674, | |
| "grad_norm": 0.27791404724121094, | |
| "learning_rate": 6.984259805044342e-05, | |
| "loss": 0.6734447479248047, | |
| "step": 1590 | |
| }, | |
| { | |
| "epoch": 0.617380714087615, | |
| "grad_norm": 0.26526719331741333, | |
| "learning_rate": 6.864304845578826e-05, | |
| "loss": 0.671996021270752, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.6212393435506626, | |
| "grad_norm": 0.28982335329055786, | |
| "learning_rate": 6.74484823300345e-05, | |
| "loss": 0.6816313743591309, | |
| "step": 1610 | |
| }, | |
| { | |
| "epoch": 0.6250979730137102, | |
| "grad_norm": 0.262482613325119, | |
| "learning_rate": 6.625908952209418e-05, | |
| "loss": 0.6751953601837158, | |
| "step": 1620 | |
| }, | |
| { | |
| "epoch": 0.6289566024767578, | |
| "grad_norm": 0.39887505769729614, | |
| "learning_rate": 6.507505905869914e-05, | |
| "loss": 0.6822636127471924, | |
| "step": 1630 | |
| }, | |
| { | |
| "epoch": 0.6328152319398054, | |
| "grad_norm": 0.2669890224933624, | |
| "learning_rate": 6.389657911435943e-05, | |
| "loss": 0.6915277957916259, | |
| "step": 1640 | |
| }, | |
| { | |
| "epoch": 0.636673861402853, | |
| "grad_norm": 0.2670292556285858, | |
| "learning_rate": 6.272383698145723e-05, | |
| "loss": 0.652358102798462, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 0.636673861402853, | |
| "eval_loss": 0.6799824237823486, | |
| "eval_runtime": 238.2058, | |
| "eval_samples_per_second": 3.505, | |
| "eval_steps_per_second": 1.755, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 0.6405324908659006, | |
| "grad_norm": 0.27272072434425354, | |
| "learning_rate": 6.15570190404811e-05, | |
| "loss": 0.6684075832366944, | |
| "step": 1660 | |
| }, | |
| { | |
| "epoch": 0.6443911203289482, | |
| "grad_norm": 0.25919416546821594, | |
| "learning_rate": 6.039631073040507e-05, | |
| "loss": 0.6651289463043213, | |
| "step": 1670 | |
| }, | |
| { | |
| "epoch": 0.6482497497919958, | |
| "grad_norm": 0.27560484409332275, | |
| "learning_rate": 5.924189651921728e-05, | |
| "loss": 0.6770682811737061, | |
| "step": 1680 | |
| }, | |
| { | |
| "epoch": 0.6521083792550434, | |
| "grad_norm": 0.29968157410621643, | |
| "learning_rate": 5.8093959874603176e-05, | |
| "loss": 0.6616378784179687, | |
| "step": 1690 | |
| }, | |
| { | |
| "epoch": 0.6598256381811385, | |
| "grad_norm": 0.2539561092853546, | |
| "learning_rate": 5.581824797953925e-05, | |
| "loss": 0.6652106761932373, | |
| "step": 1710 | |
| }, | |
| { | |
| "epoch": 0.6636842676441861, | |
| "grad_norm": 0.2645578980445862, | |
| "learning_rate": 5.4690834401347034e-05, | |
| "loss": 0.6816202163696289, | |
| "step": 1720 | |
| }, | |
| { | |
| "epoch": 0.6675428971072337, | |
| "grad_norm": 0.2860375642776489, | |
| "learning_rate": 5.357062167676426e-05, | |
| "loss": 0.6432219982147217, | |
| "step": 1730 | |
| }, | |
| { | |
| "epoch": 0.6714015265702813, | |
| "grad_norm": 0.29771652817726135, | |
| "learning_rate": 5.2457787837933715e-05, | |
| "loss": 0.6574324131011963, | |
| "step": 1740 | |
| }, | |
| { | |
| "epoch": 0.6752601560333289, | |
| "grad_norm": 0.26455527544021606, | |
| "learning_rate": 5.135250974429342e-05, | |
| "loss": 0.6606158256530762, | |
| "step": 1750 | |
| }, | |
| { | |
| "epoch": 0.6791187854963765, | |
| "grad_norm": 0.2809986174106598, | |
| "learning_rate": 5.02549630544688e-05, | |
| "loss": 0.6675248622894288, | |
| "step": 1760 | |
| }, | |
| { | |
| "epoch": 0.6829774149594241, | |
| "grad_norm": 0.2710455060005188, | |
| "learning_rate": 4.916532219835592e-05, | |
| "loss": 0.6551553249359131, | |
| "step": 1770 | |
| }, | |
| { | |
| "epoch": 0.6868360444224717, | |
| "grad_norm": 0.3084876239299774, | |
| "learning_rate": 4.808376034939965e-05, | |
| "loss": 0.6641845703125, | |
| "step": 1780 | |
| }, | |
| { | |
| "epoch": 0.6906946738855193, | |
| "grad_norm": 0.2739737033843994, | |
| "learning_rate": 4.701044939707181e-05, | |
| "loss": 0.640526533126831, | |
| "step": 1790 | |
| }, | |
| { | |
| "epoch": 0.6945533033485669, | |
| "grad_norm": 0.2665387690067291, | |
| "learning_rate": 4.594555991955327e-05, | |
| "loss": 0.6590942859649658, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.6945533033485669, | |
| "eval_loss": 0.6755020022392273, | |
| "eval_runtime": 250.8094, | |
| "eval_samples_per_second": 3.329, | |
| "eval_steps_per_second": 1.667, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.6984119328116145, | |
| "grad_norm": 0.2883066236972809, | |
| "learning_rate": 4.488926115662444e-05, | |
| "loss": 0.6711457252502442, | |
| "step": 1810 | |
| }, | |
| { | |
| "epoch": 0.7022705622746621, | |
| "grad_norm": 0.2540535628795624, | |
| "learning_rate": 4.3841720982768654e-05, | |
| "loss": 0.6684216499328614, | |
| "step": 1820 | |
| }, | |
| { | |
| "epoch": 0.7061291917377097, | |
| "grad_norm": 0.28529274463653564, | |
| "learning_rate": 4.2803105880491925e-05, | |
| "loss": 0.667764949798584, | |
| "step": 1830 | |
| }, | |
| { | |
| "epoch": 0.7099878212007572, | |
| "grad_norm": 0.2879117727279663, | |
| "learning_rate": 4.177358091386495e-05, | |
| "loss": 0.6530799865722656, | |
| "step": 1840 | |
| }, | |
| { | |
| "epoch": 0.7138464506638048, | |
| "grad_norm": 0.2830539643764496, | |
| "learning_rate": 4.075330970228948e-05, | |
| "loss": 0.6635762691497803, | |
| "step": 1850 | |
| }, | |
| { | |
| "epoch": 0.7177050801268524, | |
| "grad_norm": 0.2935732305049896, | |
| "learning_rate": 3.974245439449507e-05, | |
| "loss": 0.6464953422546387, | |
| "step": 1860 | |
| }, | |
| { | |
| "epoch": 0.7215637095899, | |
| "grad_norm": 0.2779483199119568, | |
| "learning_rate": 3.874117564276905e-05, | |
| "loss": 0.6604641914367676, | |
| "step": 1870 | |
| }, | |
| { | |
| "epoch": 0.7254223390529476, | |
| "grad_norm": 0.27317818999290466, | |
| "learning_rate": 3.774963257742462e-05, | |
| "loss": 0.6562795162200927, | |
| "step": 1880 | |
| }, | |
| { | |
| "epoch": 0.7292809685159952, | |
| "grad_norm": 0.37867748737335205, | |
| "learning_rate": 3.676798278151077e-05, | |
| "loss": 0.643172025680542, | |
| "step": 1890 | |
| }, | |
| { | |
| "epoch": 0.7331395979790428, | |
| "grad_norm": 0.2741486132144928, | |
| "learning_rate": 3.5796382265767937e-05, | |
| "loss": 0.6483676910400391, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.7369982274420904, | |
| "grad_norm": 0.2868051826953888, | |
| "learning_rate": 3.483498544383392e-05, | |
| "loss": 0.6868988990783691, | |
| "step": 1910 | |
| }, | |
| { | |
| "epoch": 0.740856856905138, | |
| "grad_norm": 0.2656330466270447, | |
| "learning_rate": 3.388394510770296e-05, | |
| "loss": 0.6492169857025146, | |
| "step": 1920 | |
| }, | |
| { | |
| "epoch": 0.7447154863681856, | |
| "grad_norm": 0.2695057988166809, | |
| "learning_rate": 3.294341240344343e-05, | |
| "loss": 0.6672848224639892, | |
| "step": 1930 | |
| }, | |
| { | |
| "epoch": 0.7485741158312332, | |
| "grad_norm": 0.30844730138778687, | |
| "learning_rate": 3.2013536807176245e-05, | |
| "loss": 0.694272804260254, | |
| "step": 1940 | |
| }, | |
| { | |
| "epoch": 0.7524327452942808, | |
| "grad_norm": 0.2679760754108429, | |
| "learning_rate": 3.1094466101319356e-05, | |
| "loss": 0.6574278354644776, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 0.7524327452942808, | |
| "eval_loss": 0.670225203037262, | |
| "eval_runtime": 239.7729, | |
| "eval_samples_per_second": 3.482, | |
| "eval_steps_per_second": 1.743, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 0.7562913747573283, | |
| "grad_norm": 0.293067991733551, | |
| "learning_rate": 3.018634635110077e-05, | |
| "loss": 0.6655418395996093, | |
| "step": 1960 | |
| }, | |
| { | |
| "epoch": 0.7601500042203759, | |
| "grad_norm": 0.2547478675842285, | |
| "learning_rate": 2.9289321881345254e-05, | |
| "loss": 0.6371743679046631, | |
| "step": 1970 | |
| }, | |
| { | |
| "epoch": 0.7640086336834235, | |
| "grad_norm": 0.28070300817489624, | |
| "learning_rate": 2.8403535253536872e-05, | |
| "loss": 0.6535021305084229, | |
| "step": 1980 | |
| }, | |
| { | |
| "epoch": 0.7678672631464711, | |
| "grad_norm": 0.27538198232650757, | |
| "learning_rate": 2.7529127243162235e-05, | |
| "loss": 0.6596447467803955, | |
| "step": 1990 | |
| } | |
| ], | |
| "logging_steps": 10, | |
| "max_steps": 2592, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": false, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.404426348587275e+19, | |
| "train_batch_size": 2, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |