Download checkpoint-700/trainer_state.json from Nihilux/SpringHunter: direct link, hf CLI and curl.
- Browser
- Download file 13.6 kB
-
https://huggingface.co/Nihilux/SpringHunter/resolve/main/checkpoint-700/trainer_state.json
- Command line
-
hf download hf://Nihilux/SpringHunter/checkpoint-700/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/Nihilux/SpringHunter/resolve/main/checkpoint-700/trainer_state.json
13.6 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.27010406241333157, | |
| "eval_steps": 150, | |
| "global_step": 700, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.003858629463047594, | |
| "grad_norm": 0.9612510204315186, | |
| "learning_rate": 1.8e-05, | |
| "loss": 1.0067429542541504, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.007717258926095188, | |
| "grad_norm": 0.6660895347595215, | |
| "learning_rate": 3.8e-05, | |
| "loss": 0.8466403007507324, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.011575888389142782, | |
| "grad_norm": 0.5681023001670837, | |
| "learning_rate": 5.8e-05, | |
| "loss": 0.7557645797729492, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.015434517852190376, | |
| "grad_norm": 0.7179134488105774, | |
| "learning_rate": 7.800000000000001e-05, | |
| "loss": 0.7409078121185303, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.019293147315237968, | |
| "grad_norm": 0.5565946102142334, | |
| "learning_rate": 9.8e-05, | |
| "loss": 0.7274651050567627, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.023151776778285563, | |
| "grad_norm": 0.5450749397277832, | |
| "learning_rate": 0.000118, | |
| "loss": 0.741389274597168, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.027010406241333156, | |
| "grad_norm": 0.4871410131454468, | |
| "learning_rate": 0.000138, | |
| "loss": 0.725303316116333, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.03086903570438075, | |
| "grad_norm": 0.5452415347099304, | |
| "learning_rate": 0.00015800000000000002, | |
| "loss": 0.724397611618042, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.03472766516742835, | |
| "grad_norm": 0.438076376914978, | |
| "learning_rate": 0.00017800000000000002, | |
| "loss": 0.7216084480285645, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.038586294630475935, | |
| "grad_norm": 0.3792989253997803, | |
| "learning_rate": 0.00019800000000000002, | |
| "loss": 0.7241204738616943, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.04244492409352353, | |
| "grad_norm": 0.31958940625190735, | |
| "learning_rate": 0.0001999935634368633, | |
| "loss": 0.7156620502471924, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.04630355355657113, | |
| "grad_norm": 0.3532995283603668, | |
| "learning_rate": 0.00019997131465276176, | |
| "loss": 0.716402530670166, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.05016218301961872, | |
| "grad_norm": 0.3720763921737671, | |
| "learning_rate": 0.0001999331777190454, | |
| "loss": 0.725121021270752, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.05402081248266631, | |
| "grad_norm": 0.327168345451355, | |
| "learning_rate": 0.0001998791586967059, | |
| "loss": 0.711461877822876, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.05787944194571391, | |
| "grad_norm": 0.31293419003486633, | |
| "learning_rate": 0.00019980926617082901, | |
| "loss": 0.7175386905670166, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.05787944194571391, | |
| "eval_loss": 0.7297696471214294, | |
| "eval_runtime": 244.1933, | |
| "eval_samples_per_second": 3.419, | |
| "eval_steps_per_second": 1.712, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.0617380714087615, | |
| "grad_norm": 0.3199326694011688, | |
| "learning_rate": 0.0001997235112492302, | |
| "loss": 0.7084239006042481, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.06559670087180909, | |
| "grad_norm": 0.3334254026412964, | |
| "learning_rate": 0.00019962190756068907, | |
| "loss": 0.7120396614074707, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.0694553303348567, | |
| "grad_norm": 0.32970502972602844, | |
| "learning_rate": 0.00019950447125278376, | |
| "loss": 0.7141448974609375, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.07331395979790428, | |
| "grad_norm": 0.3339720368385315, | |
| "learning_rate": 0.00019937122098932428, | |
| "loss": 0.7231975555419922, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.07717258926095187, | |
| "grad_norm": 0.302263081073761, | |
| "learning_rate": 0.00019922217794738657, | |
| "loss": 0.7377682685852051, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.08103121872399947, | |
| "grad_norm": 0.3148849904537201, | |
| "learning_rate": 0.00019905736581394689, | |
| "loss": 0.7064791202545166, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.08488984818704706, | |
| "grad_norm": 0.3036375641822815, | |
| "learning_rate": 0.00019887681078211707, | |
| "loss": 0.7082621097564697, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.08874847765009465, | |
| "grad_norm": 0.353236585855484, | |
| "learning_rate": 0.00019868054154698202, | |
| "loss": 0.7202902793884277, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.09260710711314225, | |
| "grad_norm": 0.2983817458152771, | |
| "learning_rate": 0.0001984685893010392, | |
| "loss": 0.6949288845062256, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.09646573657618984, | |
| "grad_norm": 0.3325659930706024, | |
| "learning_rate": 0.00019824098772924114, | |
| "loss": 0.7241635799407959, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.10032436603923744, | |
| "grad_norm": 0.29093560576438904, | |
| "learning_rate": 0.00019799777300364218, | |
| "loss": 0.7079925060272216, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.10418299550228503, | |
| "grad_norm": 0.3151165246963501, | |
| "learning_rate": 0.0001977389837776497, | |
| "loss": 0.7299640655517579, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.10804162496533262, | |
| "grad_norm": 0.30055317282676697, | |
| "learning_rate": 0.000197464661179881, | |
| "loss": 0.7072536468505859, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.11190025442838022, | |
| "grad_norm": 0.28629517555236816, | |
| "learning_rate": 0.00019717484880762685, | |
| "loss": 0.7206380844116211, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.1196175133544754, | |
| "grad_norm": 0.31153738498687744, | |
| "learning_rate": 0.0001965489414302289, | |
| "loss": 0.722527265548706, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.123476142817523, | |
| "grad_norm": 0.2931964695453644, | |
| "learning_rate": 0.00019621294589872003, | |
| "loss": 0.7037133693695068, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.1273347722805706, | |
| "grad_norm": 0.33466067910194397, | |
| "learning_rate": 0.0001958616595241865, | |
| "loss": 0.727421760559082, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.13119340174361818, | |
| "grad_norm": 0.28575319051742554, | |
| "learning_rate": 0.00019549513813554785, | |
| "loss": 0.6994576930999756, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.13505203120666578, | |
| "grad_norm": 0.31457364559173584, | |
| "learning_rate": 0.0001951134399829799, | |
| "loss": 0.734303331375122, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.1389106606697134, | |
| "grad_norm": 0.30264604091644287, | |
| "learning_rate": 0.00019471662572865736, | |
| "loss": 0.7409284591674805, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.14276929013276096, | |
| "grad_norm": 0.28342851996421814, | |
| "learning_rate": 0.00019430475843711293, | |
| "loss": 0.7335753440856934, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.14662791959580856, | |
| "grad_norm": 0.30577903985977173, | |
| "learning_rate": 0.00019387790356521463, | |
| "loss": 0.7155701637268066, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.15048654905885617, | |
| "grad_norm": 0.28058379888534546, | |
| "learning_rate": 0.000193436128951763, | |
| "loss": 0.6958878993988037, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.15434517852190374, | |
| "grad_norm": 0.30901074409484863, | |
| "learning_rate": 0.0001929795048067095, | |
| "loss": 0.7224118709564209, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.15820380798495134, | |
| "grad_norm": 0.28316858410835266, | |
| "learning_rate": 0.0001925081036999984, | |
| "loss": 0.7156408786773681, | |
| "step": 410 | |
| }, | |
| { | |
| "epoch": 0.16206243744799895, | |
| "grad_norm": 0.29417017102241516, | |
| "learning_rate": 0.00019202200055003346, | |
| "loss": 0.7048601150512696, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.16592106691104652, | |
| "grad_norm": 0.29546213150024414, | |
| "learning_rate": 0.00019152127261177126, | |
| "loss": 0.7047354698181152, | |
| "step": 430 | |
| }, | |
| { | |
| "epoch": 0.16977969637409412, | |
| "grad_norm": 0.2986455261707306, | |
| "learning_rate": 0.0001910059994644434, | |
| "loss": 0.6936575889587402, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 0.17363832583714173, | |
| "grad_norm": 0.35280054807662964, | |
| "learning_rate": 0.0001904762629989091, | |
| "loss": 0.6981217384338378, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.17363832583714173, | |
| "eval_loss": 0.7733834385871887, | |
| "eval_runtime": 241.1126, | |
| "eval_samples_per_second": 3.463, | |
| "eval_steps_per_second": 1.734, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.1774969553001893, | |
| "grad_norm": 0.28504979610443115, | |
| "learning_rate": 0.00018993214740464063, | |
| "loss": 0.7187217235565185, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 0.1813555847632369, | |
| "grad_norm": 0.294179230928421, | |
| "learning_rate": 0.00018937373915634323, | |
| "loss": 0.723856258392334, | |
| "step": 470 | |
| }, | |
| { | |
| "epoch": 0.1852142142262845, | |
| "grad_norm": 0.3039489984512329, | |
| "learning_rate": 0.00018880112700021205, | |
| "loss": 0.7158075332641601, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 0.18907284368933208, | |
| "grad_norm": 0.3352907598018646, | |
| "learning_rate": 0.0001882144019398278, | |
| "loss": 0.7123313903808594, | |
| "step": 490 | |
| }, | |
| { | |
| "epoch": 0.19293147315237968, | |
| "grad_norm": 0.31155890226364136, | |
| "learning_rate": 0.00018761365722169403, | |
| "loss": 0.7325988292694092, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.1967901026154273, | |
| "grad_norm": 0.3036295175552368, | |
| "learning_rate": 0.00018699898832041757, | |
| "loss": 0.7070968627929688, | |
| "step": 510 | |
| }, | |
| { | |
| "epoch": 0.2006487320784749, | |
| "grad_norm": 0.2792280316352844, | |
| "learning_rate": 0.00018637049292353513, | |
| "loss": 0.7023006916046143, | |
| "step": 520 | |
| }, | |
| { | |
| "epoch": 0.20450736154152246, | |
| "grad_norm": 0.30598390102386475, | |
| "learning_rate": 0.00018572827091598793, | |
| "loss": 0.708428144454956, | |
| "step": 530 | |
| }, | |
| { | |
| "epoch": 0.20836599100457007, | |
| "grad_norm": 0.28768792748451233, | |
| "learning_rate": 0.00018507242436424765, | |
| "loss": 0.708616304397583, | |
| "step": 540 | |
| }, | |
| { | |
| "epoch": 0.21222462046761767, | |
| "grad_norm": 0.32237133383750916, | |
| "learning_rate": 0.00018440305750009483, | |
| "loss": 0.7210727214813233, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 0.21608324993066524, | |
| "grad_norm": 0.2892608940601349, | |
| "learning_rate": 0.000183720276704054, | |
| "loss": 0.6931821823120117, | |
| "step": 560 | |
| }, | |
| { | |
| "epoch": 0.21994187939371285, | |
| "grad_norm": 0.3152562081813812, | |
| "learning_rate": 0.00018302419048848667, | |
| "loss": 0.7213119506835938, | |
| "step": 570 | |
| }, | |
| { | |
| "epoch": 0.22380050885676045, | |
| "grad_norm": 0.31499791145324707, | |
| "learning_rate": 0.00018231490948034592, | |
| "loss": 0.6939045906066894, | |
| "step": 580 | |
| }, | |
| { | |
| "epoch": 0.22765913831980802, | |
| "grad_norm": 0.2861924469470978, | |
| "learning_rate": 0.00018159254640359487, | |
| "loss": 0.7252971649169921, | |
| "step": 590 | |
| }, | |
| { | |
| "epoch": 0.23537639724590323, | |
| "grad_norm": 0.2836938500404358, | |
| "learning_rate": 0.00018010903531734363, | |
| "loss": 0.6902801513671875, | |
| "step": 610 | |
| }, | |
| { | |
| "epoch": 0.2392350267089508, | |
| "grad_norm": 0.3197932243347168, | |
| "learning_rate": 0.00017934812307793583, | |
| "loss": 0.6933451175689698, | |
| "step": 620 | |
| }, | |
| { | |
| "epoch": 0.2430936561719984, | |
| "grad_norm": 0.29024383425712585, | |
| "learning_rate": 0.00017857460027263225, | |
| "loss": 0.6954158306121826, | |
| "step": 630 | |
| }, | |
| { | |
| "epoch": 0.246952285635046, | |
| "grad_norm": 0.29053837060928345, | |
| "learning_rate": 0.00017778858983515743, | |
| "loss": 0.7062996864318848, | |
| "step": 640 | |
| }, | |
| { | |
| "epoch": 0.2508109150980936, | |
| "grad_norm": 0.2862823009490967, | |
| "learning_rate": 0.00017699021668385895, | |
| "loss": 0.726755428314209, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 0.2546695445611412, | |
| "grad_norm": 0.31322675943374634, | |
| "learning_rate": 0.00017617960770185444, | |
| "loss": 0.7403687953948974, | |
| "step": 660 | |
| }, | |
| { | |
| "epoch": 0.25852817402418876, | |
| "grad_norm": 0.2916722595691681, | |
| "learning_rate": 0.00017535689171686644, | |
| "loss": 0.6931378364562988, | |
| "step": 670 | |
| }, | |
| { | |
| "epoch": 0.26238680348723636, | |
| "grad_norm": 0.3099428117275238, | |
| "learning_rate": 0.00017452219948074814, | |
| "loss": 0.7085556983947754, | |
| "step": 680 | |
| }, | |
| { | |
| "epoch": 0.26624543295028397, | |
| "grad_norm": 0.30417969822883606, | |
| "learning_rate": 0.0001736756636487035, | |
| "loss": 0.7230380058288575, | |
| "step": 690 | |
| } | |
| ], | |
| "logging_steps": 10, | |
| "max_steps": 2592, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": false, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 8.261857846271484e+18, | |
| "train_batch_size": 2, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |