Instructions to use CodeIsAbstract/HybridTimeScaleModel_conti4 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use CodeIsAbstract/HybridTimeScaleModel_conti4 with Transformers:
# Load model directly from transformers import HybridTimeScaleLM model = HybridTimeScaleLM.from_pretrained("CodeIsAbstract/HybridTimeScaleModel_conti4", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from CodeIsAbstract/HybridTimeScaleModel_conti4: direct link, hf CLI and curl.
- Browser
- Download file 7.31 kB
-
https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti4/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/HybridTimeScaleModel_conti4/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti4/resolve/main/last-checkpoint/trainer_state.json
7.31 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.4, | |
| "eval_steps": 50, | |
| "global_step": 200, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": 2880.0, | |
| "learning_rate": 7.999999999999999e-05, | |
| "loss": 3.1943845748901367, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": 111104.0, | |
| "learning_rate": 0.00017999999999999998, | |
| "loss": 3.2289222717285155, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": 7776.0, | |
| "learning_rate": 0.00028, | |
| "loss": 3.2137096405029295, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": 13952.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.192697525024414, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 34304.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.1834150314331056, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 3296.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2362594604492188, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 33024.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2107208251953123, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "grad_norm": 45312.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.219687652587891, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.09, | |
| "grad_norm": 2112.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2011333465576173, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": 3584.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2285308837890625, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "eval_loss": 3.5073623657226562, | |
| "eval_runtime": 9.0594, | |
| "eval_samples_per_second": 1.987, | |
| "eval_steps_per_second": 1.987, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.11, | |
| "grad_norm": 828.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.205875778198242, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "grad_norm": 8096.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2160839080810546, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.13, | |
| "grad_norm": 9728.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2235694885253907, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "grad_norm": 35584.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.271558380126953, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "grad_norm": 2672.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.263863372802734, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "grad_norm": 5504.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.264923858642578, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.17, | |
| "grad_norm": 296960.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2530628204345704, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.18, | |
| "grad_norm": 2608.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.279144287109375, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.19, | |
| "grad_norm": 1736.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2283939361572265, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "grad_norm": 2928.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.276975631713867, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "eval_loss": 3.581308126449585, | |
| "eval_runtime": 9.0248, | |
| "eval_samples_per_second": 1.995, | |
| "eval_steps_per_second": 1.995, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.21, | |
| "grad_norm": 2384.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.2437297821044924, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.22, | |
| "grad_norm": 13440.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.1839040756225585, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.23, | |
| "grad_norm": 744.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.157756805419922, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.24, | |
| "grad_norm": 207.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.179922676086426, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.25, | |
| "grad_norm": 660.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.1574981689453123, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.26, | |
| "grad_norm": 99.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.1093936920166017, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.27, | |
| "grad_norm": 936.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.068846893310547, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.28, | |
| "grad_norm": 1296.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.043972206115723, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.29, | |
| "grad_norm": 152.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.063678741455078, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "grad_norm": 50.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.0807422637939452, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "eval_loss": 3.381226062774658, | |
| "eval_runtime": 8.9995, | |
| "eval_samples_per_second": 2.0, | |
| "eval_steps_per_second": 2.0, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.31, | |
| "grad_norm": 652.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.050394821166992, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.32, | |
| "grad_norm": 187.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.1013137817382814, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.33, | |
| "grad_norm": 430.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.052425003051758, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.34, | |
| "grad_norm": 147.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.0121395111083986, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.35, | |
| "grad_norm": 382.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.015604591369629, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.36, | |
| "grad_norm": 368.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.006996726989746, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.37, | |
| "grad_norm": 14.5625, | |
| "learning_rate": 0.0003, | |
| "loss": 3.026082992553711, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.38, | |
| "grad_norm": 158.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.029666709899902, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.39, | |
| "grad_norm": 62.25, | |
| "learning_rate": 0.0003, | |
| "loss": 2.9676319122314454, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.4, | |
| "grad_norm": 1496.0, | |
| "learning_rate": 0.0003, | |
| "loss": 3.016917419433594, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.4, | |
| "eval_loss": 3.3380279541015625, | |
| "eval_runtime": 9.0151, | |
| "eval_samples_per_second": 1.997, | |
| "eval_steps_per_second": 1.997, | |
| "step": 200 | |
| } | |
| ], | |
| "logging_steps": 5, | |
| "max_steps": 500, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 200, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 5.79830248636416e+16, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |