Instructions to use CodeIsAbstract/HybridTimeScaleModel_conti3_ with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use CodeIsAbstract/HybridTimeScaleModel_conti3_ with Transformers:
# Load model directly from transformers import HybridTimeScaleLM model = HybridTimeScaleLM.from_pretrained("CodeIsAbstract/HybridTimeScaleModel_conti3_", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from CodeIsAbstract/HybridTimeScaleModel_conti3_: direct link, hf CLI and curl.
- Browser
- Download file 7.42 kB
-
https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti3_/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/HybridTimeScaleModel_conti3_/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/HybridTimeScaleModel_conti3_/resolve/main/last-checkpoint/trainer_state.json
7.42 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.2, | |
| "eval_steps": 50, | |
| "global_step": 200, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.005, | |
| "grad_norm": 1.859375, | |
| "learning_rate": 7.999999999999999e-05, | |
| "loss": 2.9596134185791017, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": 2.375, | |
| "learning_rate": 0.00017999999999999998, | |
| "loss": 3.0003786087036133, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.015, | |
| "grad_norm": 14.75, | |
| "learning_rate": 0.00028, | |
| "loss": 2.946580696105957, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": 1.2421875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.9287084579467773, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.025, | |
| "grad_norm": 3.296875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8551090240478514, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": 1.5546875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8845558166503906, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.035, | |
| "grad_norm": 1.03125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8326595306396483, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": 5.03125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.844678497314453, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.045, | |
| "grad_norm": 2.046875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.828007125854492, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 9.25, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8197803497314453, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "eval_loss": 3.1455771923065186, | |
| "eval_runtime": 6.8102, | |
| "eval_samples_per_second": 5.433, | |
| "eval_steps_per_second": 2.79, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.055, | |
| "grad_norm": 1.6171875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.797296714782715, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 1.90625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7909267425537108, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.065, | |
| "grad_norm": 0.66796875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7826940536499025, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 1.1640625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8065961837768554, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.075, | |
| "grad_norm": 0.61328125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.8089221954345702, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "grad_norm": 2.21875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7947248458862304, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.085, | |
| "grad_norm": 0.7890625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7826770782470702, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.09, | |
| "grad_norm": 26.5, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7801084518432617, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.095, | |
| "grad_norm": 1.2734375, | |
| "learning_rate": 0.0003, | |
| "loss": 2.748321533203125, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": 0.341796875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7878595352172852, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "eval_loss": 3.1193559169769287, | |
| "eval_runtime": 6.6531, | |
| "eval_samples_per_second": 5.561, | |
| "eval_steps_per_second": 2.856, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.105, | |
| "grad_norm": 1.078125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.785268211364746, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.11, | |
| "grad_norm": 0.890625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.725397491455078, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.115, | |
| "grad_norm": 2.84375, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7376623153686523, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "grad_norm": 2.421875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.758837127685547, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.125, | |
| "grad_norm": 1.828125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.768202018737793, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.13, | |
| "grad_norm": 0.3515625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7715227127075197, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.135, | |
| "grad_norm": 0.5, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7676155090332033, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "grad_norm": 0.83203125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.739817428588867, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.145, | |
| "grad_norm": 0.66796875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.740823745727539, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "grad_norm": 12.5, | |
| "learning_rate": 0.0003, | |
| "loss": 2.754258728027344, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "eval_loss": 3.1168980598449707, | |
| "eval_runtime": 6.6806, | |
| "eval_samples_per_second": 5.538, | |
| "eval_steps_per_second": 2.844, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.155, | |
| "grad_norm": 2.359375, | |
| "learning_rate": 0.0003, | |
| "loss": 2.713482475280762, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "grad_norm": 2.265625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7810474395751954, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.165, | |
| "grad_norm": 0.5625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.758929443359375, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.17, | |
| "grad_norm": 1.6015625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7030134201049805, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.175, | |
| "grad_norm": 0.74609375, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7331872940063477, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.18, | |
| "grad_norm": 2.5, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7181303024291994, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.185, | |
| "grad_norm": 0.53515625, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7308780670166017, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.19, | |
| "grad_norm": 3.875, | |
| "learning_rate": 0.0003, | |
| "loss": 2.7536849975585938, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.195, | |
| "grad_norm": 0.361328125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.71515007019043, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "grad_norm": 0.408203125, | |
| "learning_rate": 0.0003, | |
| "loss": 2.712385559082031, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "eval_loss": 3.117068290710449, | |
| "eval_runtime": 6.7047, | |
| "eval_samples_per_second": 5.519, | |
| "eval_steps_per_second": 2.834, | |
| "step": 200 | |
| } | |
| ], | |
| "logging_steps": 5, | |
| "max_steps": 1000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 200, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 5.79830248636416e+16, | |
| "train_batch_size": 2, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |