Download last-checkpoint/trainer_state.json from CodeIsAbstract/HybridTimeScaleModel_modded: direct link, hf CLI and curl.
- Browser
- Download file 16 kB
-
https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_modded/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/HybridTimeScaleModel_modded/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_modded/resolve/main/last-checkpoint/trainer_state.json
16 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.4, | |
| "eval_steps": 50, | |
| "global_step": 400, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.005, | |
| "grad_norm": 68.2828140258789, | |
| "learning_rate": 3.2000000000000005e-05, | |
| "loss": 3.547993850708008, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": 381.42022705078125, | |
| "learning_rate": 7.2e-05, | |
| "loss": 3.2959220886230467, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.015, | |
| "grad_norm": 19640.955078125, | |
| "learning_rate": 0.00011200000000000001, | |
| "loss": 3.369282531738281, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": 205.24681091308594, | |
| "learning_rate": 0.000152, | |
| "loss": 3.256881332397461, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.025, | |
| "grad_norm": 259.6229248046875, | |
| "learning_rate": 0.000192, | |
| "loss": 3.3879322052001952, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": 2096.502685546875, | |
| "learning_rate": 0.0001999916943334945, | |
| "loss": 3.4791305541992186, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.035, | |
| "grad_norm": 29.51216697692871, | |
| "learning_rate": 0.0001999579549278937, | |
| "loss": 3.5544124603271485, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": 2868.5859375, | |
| "learning_rate": 0.00019989827142936862, | |
| "loss": 3.4086448669433596, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.045, | |
| "grad_norm": 116.66321563720703, | |
| "learning_rate": 0.00019981265932877488, | |
| "loss": 3.26219482421875, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 1199.3328857421875, | |
| "learning_rate": 0.00019970114084673796, | |
| "loss": 3.556040954589844, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "eval_loss": 3.7243571281433105, | |
| "eval_runtime": 200.7945, | |
| "eval_samples_per_second": 1.484, | |
| "eval_steps_per_second": 0.299, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.055, | |
| "grad_norm": 1653.0960693359375, | |
| "learning_rate": 0.0001995637449278864, | |
| "loss": 3.860651397705078, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 411.4178466796875, | |
| "learning_rate": 0.00019940050723333866, | |
| "loss": 3.699799728393555, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.065, | |
| "grad_norm": 2424.376708984375, | |
| "learning_rate": 0.0001992114701314478, | |
| "loss": 3.703348159790039, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 23206.84375, | |
| "learning_rate": 0.0001989966826868044, | |
| "loss": 3.6570384979248045, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.075, | |
| "grad_norm": 395.9013977050781, | |
| "learning_rate": 0.00019875620064750202, | |
| "loss": 3.622447204589844, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "grad_norm": 228293.453125, | |
| "learning_rate": 0.00019849008643066772, | |
| "loss": 3.773283767700195, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.085, | |
| "grad_norm": 52893.1875, | |
| "learning_rate": 0.00019819840910626174, | |
| "loss": 3.650778961181641, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.09, | |
| "grad_norm": 12679.291015625, | |
| "learning_rate": 0.0001978812443791503, | |
| "loss": 3.9305484771728514, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.095, | |
| "grad_norm": 439037.4375, | |
| "learning_rate": 0.0001975386745694565, | |
| "loss": 3.8479537963867188, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": 878.3585815429688, | |
| "learning_rate": 0.0001971707885911941, | |
| "loss": 3.90667724609375, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "eval_loss": 4.1160054206848145, | |
| "eval_runtime": 201.0129, | |
| "eval_samples_per_second": 1.482, | |
| "eval_steps_per_second": 0.298, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.105, | |
| "grad_norm": 2385.37744140625, | |
| "learning_rate": 0.00019677768192918971, | |
| "loss": 3.805155944824219, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.11, | |
| "grad_norm": 10.20463752746582, | |
| "learning_rate": 0.00019635945661430006, | |
| "loss": 3.822935104370117, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.115, | |
| "grad_norm": 7.702394962310791, | |
| "learning_rate": 0.0001959162211969295, | |
| "loss": 3.702631378173828, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "grad_norm": 2.8037490844726562, | |
| "learning_rate": 0.00019544809071885604, | |
| "loss": 3.6845611572265624, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.125, | |
| "grad_norm": 3.6743063926696777, | |
| "learning_rate": 0.00019495518668337201, | |
| "loss": 3.644290542602539, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.13, | |
| "grad_norm": 5.9780964851379395, | |
| "learning_rate": 0.00019443763702374812, | |
| "loss": 3.5586227416992187, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.135, | |
| "grad_norm": 2.230492353439331, | |
| "learning_rate": 0.00019389557607002805, | |
| "loss": 3.4619365692138673, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "grad_norm": 1.7121639251708984, | |
| "learning_rate": 0.00019332914451416347, | |
| "loss": 3.531117248535156, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.145, | |
| "grad_norm": 4.843437194824219, | |
| "learning_rate": 0.0001927384893734971, | |
| "loss": 3.5250553131103515, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "grad_norm": 1.5022116899490356, | |
| "learning_rate": 0.00019212376395260448, | |
| "loss": 3.252106475830078, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "eval_loss": 3.604867935180664, | |
| "eval_runtime": 200.7401, | |
| "eval_samples_per_second": 1.485, | |
| "eval_steps_per_second": 0.299, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.155, | |
| "grad_norm": 2.400775909423828, | |
| "learning_rate": 0.00019148512780350384, | |
| "loss": 3.291422653198242, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "grad_norm": 2.1534276008605957, | |
| "learning_rate": 0.00019082274668424422, | |
| "loss": 3.4382129669189454, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.165, | |
| "grad_norm": 5.130342960357666, | |
| "learning_rate": 0.00019013679251588303, | |
| "loss": 3.4447181701660154, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.17, | |
| "grad_norm": 125.57601928710938, | |
| "learning_rate": 0.00018942744333786397, | |
| "loss": 3.4935020446777343, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.175, | |
| "grad_norm": 7.2148356437683105, | |
| "learning_rate": 0.00018869488326180679, | |
| "loss": 3.4430862426757813, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.18, | |
| "grad_norm": 47.17856979370117, | |
| "learning_rate": 0.0001879393024237212, | |
| "loss": 3.4849868774414063, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.185, | |
| "grad_norm": 3.984619140625, | |
| "learning_rate": 0.00018716089693465696, | |
| "loss": 3.467825698852539, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.19, | |
| "grad_norm": 1.7979289293289185, | |
| "learning_rate": 0.00018635986882980325, | |
| "loss": 3.2273773193359374, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.195, | |
| "grad_norm": 49.847412109375, | |
| "learning_rate": 0.00018553642601605068, | |
| "loss": 3.51593017578125, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "grad_norm": 1.552674412727356, | |
| "learning_rate": 0.0001846907822180286, | |
| "loss": 3.5006553649902346, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "eval_loss": 3.619206190109253, | |
| "eval_runtime": 201.0123, | |
| "eval_samples_per_second": 1.482, | |
| "eval_steps_per_second": 0.298, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.205, | |
| "grad_norm": 5811.900390625, | |
| "learning_rate": 0.00018382315692263323, | |
| "loss": 3.4652740478515627, | |
| "step": 205 | |
| }, | |
| { | |
| "epoch": 0.21, | |
| "grad_norm": 1784.0272216796875, | |
| "learning_rate": 0.00018293377532205968, | |
| "loss": 3.136703681945801, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.215, | |
| "grad_norm": 8352.1416015625, | |
| "learning_rate": 0.0001820228682553533, | |
| "loss": 3.3410572052001952, | |
| "step": 215 | |
| }, | |
| { | |
| "epoch": 0.22, | |
| "grad_norm": 9626.669921875, | |
| "learning_rate": 0.00018109067214849538, | |
| "loss": 3.3538772583007814, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.225, | |
| "grad_norm": 5929.7529296875, | |
| "learning_rate": 0.00018013742895303883, | |
| "loss": 3.598949432373047, | |
| "step": 225 | |
| }, | |
| { | |
| "epoch": 0.23, | |
| "grad_norm": 280.86492919921875, | |
| "learning_rate": 0.0001791633860833096, | |
| "loss": 3.4480300903320313, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.235, | |
| "grad_norm": 19.305431365966797, | |
| "learning_rate": 0.00017816879635219028, | |
| "loss": 3.521527099609375, | |
| "step": 235 | |
| }, | |
| { | |
| "epoch": 0.24, | |
| "grad_norm": 11.262675285339355, | |
| "learning_rate": 0.00017715391790550252, | |
| "loss": 3.543391799926758, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.245, | |
| "grad_norm": 107.26178741455078, | |
| "learning_rate": 0.00017611901415500535, | |
| "loss": 3.380891799926758, | |
| "step": 245 | |
| }, | |
| { | |
| "epoch": 0.25, | |
| "grad_norm": 0.6105318069458008, | |
| "learning_rate": 0.00017506435371002633, | |
| "loss": 3.2983978271484373, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.25, | |
| "eval_loss": 3.6096248626708984, | |
| "eval_runtime": 200.7485, | |
| "eval_samples_per_second": 1.484, | |
| "eval_steps_per_second": 0.299, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.255, | |
| "grad_norm": 1.168889045715332, | |
| "learning_rate": 0.00017399021030774442, | |
| "loss": 3.5202346801757813, | |
| "step": 255 | |
| }, | |
| { | |
| "epoch": 0.26, | |
| "grad_norm": 4.2245283126831055, | |
| "learning_rate": 0.00017289686274214118, | |
| "loss": 3.2942405700683595, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.265, | |
| "grad_norm": 0.3621503710746765, | |
| "learning_rate": 0.00017178459479163976, | |
| "loss": 3.4601863861083983, | |
| "step": 265 | |
| }, | |
| { | |
| "epoch": 0.27, | |
| "grad_norm": 35.780120849609375, | |
| "learning_rate": 0.00017065369514545053, | |
| "loss": 3.2967288970947264, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.275, | |
| "grad_norm": 474.76861572265625, | |
| "learning_rate": 0.00016950445732864127, | |
| "loss": 3.4325428009033203, | |
| "step": 275 | |
| }, | |
| { | |
| "epoch": 0.28, | |
| "grad_norm": 2.8445627689361572, | |
| "learning_rate": 0.00016833717962595326, | |
| "loss": 3.3004375457763673, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.285, | |
| "grad_norm": 19.516523361206055, | |
| "learning_rate": 0.00016715216500438093, | |
| "loss": 3.612904739379883, | |
| "step": 285 | |
| }, | |
| { | |
| "epoch": 0.29, | |
| "grad_norm": 25.849435806274414, | |
| "learning_rate": 0.00016594972103453726, | |
| "loss": 3.567874526977539, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.295, | |
| "grad_norm": 58.06301498413086, | |
| "learning_rate": 0.00016473015981082338, | |
| "loss": 3.525835418701172, | |
| "step": 295 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "grad_norm": 191.27548217773438, | |
| "learning_rate": 0.00016349379787042477, | |
| "loss": 3.3833484649658203, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "eval_loss": 3.7416040897369385, | |
| "eval_runtime": 200.7531, | |
| "eval_samples_per_second": 1.484, | |
| "eval_steps_per_second": 0.299, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.305, | |
| "grad_norm": 56.349708557128906, | |
| "learning_rate": 0.00016224095611115384, | |
| "loss": 3.6572975158691405, | |
| "step": 305 | |
| }, | |
| { | |
| "epoch": 0.31, | |
| "grad_norm": 1745.9088134765625, | |
| "learning_rate": 0.00016097195970816094, | |
| "loss": 3.601942443847656, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.315, | |
| "grad_norm": 1165.1402587890625, | |
| "learning_rate": 0.0001596871380295351, | |
| "loss": 3.6535247802734374, | |
| "step": 315 | |
| }, | |
| { | |
| "epoch": 0.32, | |
| "grad_norm": 46.510684967041016, | |
| "learning_rate": 0.00015838682455081657, | |
| "loss": 3.5322292327880858, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.325, | |
| "grad_norm": 147.2434539794922, | |
| "learning_rate": 0.0001570713567684432, | |
| "loss": 3.4837066650390627, | |
| "step": 325 | |
| }, | |
| { | |
| "epoch": 0.33, | |
| "grad_norm": 7.8114013671875, | |
| "learning_rate": 0.00015574107611215319, | |
| "loss": 3.5842933654785156, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.335, | |
| "grad_norm": 40.32650375366211, | |
| "learning_rate": 0.00015439632785636706, | |
| "loss": 3.451204299926758, | |
| "step": 335 | |
| }, | |
| { | |
| "epoch": 0.34, | |
| "grad_norm": 54.91672134399414, | |
| "learning_rate": 0.00015303746103057162, | |
| "loss": 3.6410243988037108, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.345, | |
| "grad_norm": 143.5968017578125, | |
| "learning_rate": 0.00015166482832872923, | |
| "loss": 3.567627716064453, | |
| "step": 345 | |
| }, | |
| { | |
| "epoch": 0.35, | |
| "grad_norm": 66.56953430175781, | |
| "learning_rate": 0.00015027878601773633, | |
| "loss": 3.7443378448486326, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.35, | |
| "eval_loss": 3.8730692863464355, | |
| "eval_runtime": 200.7554, | |
| "eval_samples_per_second": 1.484, | |
| "eval_steps_per_second": 0.299, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.355, | |
| "grad_norm": 109.56166076660156, | |
| "learning_rate": 0.00014887969384495402, | |
| "loss": 3.7462512969970705, | |
| "step": 355 | |
| }, | |
| { | |
| "epoch": 0.36, | |
| "grad_norm": 403.760986328125, | |
| "learning_rate": 0.00014746791494483583, | |
| "loss": 3.4167720794677736, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.365, | |
| "grad_norm": 78.5467300415039, | |
| "learning_rate": 0.00014604381574467615, | |
| "loss": 3.653825378417969, | |
| "step": 365 | |
| }, | |
| { | |
| "epoch": 0.37, | |
| "grad_norm": 7.657294273376465, | |
| "learning_rate": 0.00014460776586950393, | |
| "loss": 3.5837642669677736, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.375, | |
| "grad_norm": 232.22654724121094, | |
| "learning_rate": 0.00014316013804614643, | |
| "loss": 3.5867897033691407, | |
| "step": 375 | |
| }, | |
| { | |
| "epoch": 0.38, | |
| "grad_norm": 34.94912338256836, | |
| "learning_rate": 0.00014170130800648814, | |
| "loss": 3.5861190795898437, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.385, | |
| "grad_norm": 91.70198059082031, | |
| "learning_rate": 0.0001402316543899493, | |
| "loss": 3.66529541015625, | |
| "step": 385 | |
| }, | |
| { | |
| "epoch": 0.39, | |
| "grad_norm": 25.11287498474121, | |
| "learning_rate": 0.0001387515586452103, | |
| "loss": 3.6498741149902343, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.395, | |
| "grad_norm": 93.66793060302734, | |
| "learning_rate": 0.0001372614049312064, | |
| "loss": 3.352254867553711, | |
| "step": 395 | |
| }, | |
| { | |
| "epoch": 0.4, | |
| "grad_norm": 4.588032245635986, | |
| "learning_rate": 0.00013576158001741932, | |
| "loss": 3.517344665527344, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.4, | |
| "eval_loss": 3.753708839416504, | |
| "eval_runtime": 200.6183, | |
| "eval_samples_per_second": 1.485, | |
| "eval_steps_per_second": 0.299, | |
| "step": 400 | |
| } | |
| ], | |
| "logging_steps": 5, | |
| "max_steps": 1000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 200, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1904846517043200.0, | |
| "train_batch_size": 2, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |