Download last-checkpoint/trainer_state.json from CodeIsAbstract/HybridTimeScaleModel_conti_ltTest: direct link, hf CLI and curl.
- Browser
- Download file 10.5 kB
-
https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti_ltTest/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/HybridTimeScaleModel_conti_ltTest/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti_ltTest/resolve/main/last-checkpoint/trainer_state.json
10.5 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.07142857142857142, | |
| "eval_steps": 100, | |
| "global_step": 500, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0014285714285714286, | |
| "grad_norm": 160.0, | |
| "learning_rate": 5.399999999999999e-05, | |
| "loss": 3.334370803833008, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.002857142857142857, | |
| "grad_norm": 143.0, | |
| "learning_rate": 0.00011399999999999999, | |
| "loss": 3.312285232543945, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.004285714285714286, | |
| "grad_norm": 91.0, | |
| "learning_rate": 0.00017399999999999997, | |
| "loss": 3.3183692932128905, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.005714285714285714, | |
| "grad_norm": 306.0, | |
| "learning_rate": 0.000234, | |
| "loss": 3.232763671875, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.007142857142857143, | |
| "grad_norm": 114.5, | |
| "learning_rate": 0.000294, | |
| "loss": 3.1770435333251954, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.008571428571428572, | |
| "grad_norm": 148.0, | |
| "learning_rate": 0.00029999875870267496, | |
| "loss": 3.1800256729125977, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": 25.0, | |
| "learning_rate": 0.0002999944678247173, | |
| "loss": 3.166558837890625, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.011428571428571429, | |
| "grad_norm": 22.75, | |
| "learning_rate": 0.0002999871121291227, | |
| "loss": 3.203672409057617, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.012857142857142857, | |
| "grad_norm": 6.4375, | |
| "learning_rate": 0.00029997669176618923, | |
| "loss": 3.155362129211426, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.014285714285714285, | |
| "grad_norm": 142.0, | |
| "learning_rate": 0.0002999632069488347, | |
| "loss": 3.1418533325195312, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.014285714285714285, | |
| "eval_loss": 3.406630277633667, | |
| "eval_runtime": 4.5564, | |
| "eval_samples_per_second": 65.402, | |
| "eval_steps_per_second": 3.731, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.015714285714285715, | |
| "grad_norm": 200.0, | |
| "learning_rate": 0.00029994665795259274, | |
| "loss": 3.106761932373047, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.017142857142857144, | |
| "grad_norm": 193.0, | |
| "learning_rate": 0.0002999270451156068, | |
| "loss": 3.1113397598266603, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.018571428571428572, | |
| "grad_norm": 39.5, | |
| "learning_rate": 0.0002999043688386235, | |
| "loss": 3.0628725051879884, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": 3.125, | |
| "learning_rate": 0.0002998786295849843, | |
| "loss": 3.030979347229004, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.02142857142857143, | |
| "grad_norm": 10.0625, | |
| "learning_rate": 0.000299849827880616, | |
| "loss": 3.0148197174072267, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.022857142857142857, | |
| "grad_norm": 4.03125, | |
| "learning_rate": 0.00029981796431402015, | |
| "loss": 2.963352394104004, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.024285714285714285, | |
| "grad_norm": 28.5, | |
| "learning_rate": 0.0002997830395362608, | |
| "loss": 2.951630401611328, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.025714285714285714, | |
| "grad_norm": 5.15625, | |
| "learning_rate": 0.00029974505426095166, | |
| "loss": 2.903683662414551, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.027142857142857142, | |
| "grad_norm": 2.359375, | |
| "learning_rate": 0.0002997040092642407, | |
| "loss": 2.911179542541504, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.02857142857142857, | |
| "grad_norm": 1.484375, | |
| "learning_rate": 0.00029965990538479526, | |
| "loss": 2.8417181015014648, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.02857142857142857, | |
| "eval_loss": 3.159050703048706, | |
| "eval_runtime": 4.4564, | |
| "eval_samples_per_second": 66.87, | |
| "eval_steps_per_second": 3.815, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": 1.515625, | |
| "learning_rate": 0.0002996127435237841, | |
| "loss": 2.815742301940918, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.03142857142857143, | |
| "grad_norm": 1.71875, | |
| "learning_rate": 0.0002995625246448595, | |
| "loss": 2.7929365158081056, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.032857142857142856, | |
| "grad_norm": 0.765625, | |
| "learning_rate": 0.00029950924977413735, | |
| "loss": 2.7637331008911135, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.03428571428571429, | |
| "grad_norm": 0.609375, | |
| "learning_rate": 0.0002994529200001762, | |
| "loss": 2.73776912689209, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.03571428571428571, | |
| "grad_norm": 1.5703125, | |
| "learning_rate": 0.00029939353647395506, | |
| "loss": 2.723506736755371, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.037142857142857144, | |
| "grad_norm": 0.66015625, | |
| "learning_rate": 0.00029933110040884987, | |
| "loss": 2.6915863037109373, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.03857142857142857, | |
| "grad_norm": 0.9609375, | |
| "learning_rate": 0.00029926561308060874, | |
| "loss": 2.663774871826172, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": 0.65234375, | |
| "learning_rate": 0.00029919707582732577, | |
| "loss": 2.6552204132080077, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.041428571428571426, | |
| "grad_norm": 0.1708984375, | |
| "learning_rate": 0.0002991254900494139, | |
| "loss": 2.6489795684814452, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.04285714285714286, | |
| "grad_norm": 0.984375, | |
| "learning_rate": 0.000299050857209576, | |
| "loss": 2.6665163040161133, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.04285714285714286, | |
| "eval_loss": 3.0404820442199707, | |
| "eval_runtime": 4.4652, | |
| "eval_samples_per_second": 66.739, | |
| "eval_steps_per_second": 3.807, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.04428571428571428, | |
| "grad_norm": 1.2890625, | |
| "learning_rate": 0.00029897317883277537, | |
| "loss": 2.6572853088378907, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.045714285714285714, | |
| "grad_norm": 0.2265625, | |
| "learning_rate": 0.00029889245650620413, | |
| "loss": 2.642633056640625, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.047142857142857146, | |
| "grad_norm": 0.6484375, | |
| "learning_rate": 0.0002988086918792514, | |
| "loss": 2.6547801971435545, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.04857142857142857, | |
| "grad_norm": 4.625, | |
| "learning_rate": 0.0002987218866634688, | |
| "loss": 2.651763153076172, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 0.2275390625, | |
| "learning_rate": 0.00029863204263253624, | |
| "loss": 2.635426902770996, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.05142857142857143, | |
| "grad_norm": 0.11328125, | |
| "learning_rate": 0.0002985391616222252, | |
| "loss": 2.6323591232299806, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.05285714285714286, | |
| "grad_norm": 3.125, | |
| "learning_rate": 0.0002984432455303614, | |
| "loss": 2.655726432800293, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.054285714285714284, | |
| "grad_norm": 0.3359375, | |
| "learning_rate": 0.00029834429631678597, | |
| "loss": 2.6482254028320313, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.055714285714285716, | |
| "grad_norm": 0.54296875, | |
| "learning_rate": 0.00029824231600331547, | |
| "loss": 2.6480077743530273, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.05714285714285714, | |
| "grad_norm": 0.166015625, | |
| "learning_rate": 0.0002981373066737005, | |
| "loss": 2.6565391540527346, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.05714285714285714, | |
| "eval_loss": 3.0312750339508057, | |
| "eval_runtime": 4.4433, | |
| "eval_samples_per_second": 67.068, | |
| "eval_steps_per_second": 3.826, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.05857142857142857, | |
| "grad_norm": 0.255859375, | |
| "learning_rate": 0.0002980292704735831, | |
| "loss": 2.663307762145996, | |
| "step": 410 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 0.0966796875, | |
| "learning_rate": 0.00029791820961045317, | |
| "loss": 2.6410280227661134, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.06142857142857143, | |
| "grad_norm": 0.3125, | |
| "learning_rate": 0.0002978041263536029, | |
| "loss": 2.6257579803466795, | |
| "step": 430 | |
| }, | |
| { | |
| "epoch": 0.06285714285714286, | |
| "grad_norm": 0.310546875, | |
| "learning_rate": 0.0002976870230340808, | |
| "loss": 2.650602912902832, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 0.06428571428571428, | |
| "grad_norm": 0.115234375, | |
| "learning_rate": 0.0002975669020446439, | |
| "loss": 2.6274070739746094, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.06571428571428571, | |
| "grad_norm": 0.10693359375, | |
| "learning_rate": 0.00029744376583970897, | |
| "loss": 2.629538154602051, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 0.06714285714285714, | |
| "grad_norm": 0.111328125, | |
| "learning_rate": 0.0002973176169353022, | |
| "loss": 2.6362348556518556, | |
| "step": 470 | |
| }, | |
| { | |
| "epoch": 0.06857142857142857, | |
| "grad_norm": 1.1015625, | |
| "learning_rate": 0.00029718845790900785, | |
| "loss": 2.613158416748047, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 0.412109375, | |
| "learning_rate": 0.00029705629139991567, | |
| "loss": 2.658156967163086, | |
| "step": 490 | |
| }, | |
| { | |
| "epoch": 0.07142857142857142, | |
| "grad_norm": 0.42578125, | |
| "learning_rate": 0.000296921120108567, | |
| "loss": 2.6301057815551756, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.07142857142857142, | |
| "eval_loss": 3.0305709838867188, | |
| "eval_runtime": 4.4742, | |
| "eval_samples_per_second": 66.604, | |
| "eval_steps_per_second": 3.8, | |
| "step": 500 | |
| } | |
| ], | |
| "logging_steps": 10, | |
| "max_steps": 7000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.63077257428992e+17, | |
| "train_batch_size": 18, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |