{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 500, "global_step": 600, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.125, "grad_norm": 0.2043563425540924, "learning_rate": 0.00019994755690455152, "loss": 0.2003, "step": 25 }, { "epoch": 0.25, "grad_norm": 0.16403928399085999, "learning_rate": 0.00019860320222975431, "loss": 0.1158, "step": 50 }, { "epoch": 0.375, "grad_norm": 0.2802892327308655, "learning_rate": 0.00019546590805746052, "loss": 0.1094, "step": 75 }, { "epoch": 0.5, "grad_norm": 0.17004738748073578, "learning_rate": 0.0001905927209998447, "loss": 0.1016, "step": 100 }, { "epoch": 0.625, "grad_norm": 0.14632447063922882, "learning_rate": 0.00018407225206909208, "loss": 0.1038, "step": 125 }, { "epoch": 0.75, "grad_norm": 0.21698755025863647, "learning_rate": 0.00017602306542969005, "loss": 0.101, "step": 150 }, { "epoch": 0.875, "grad_norm": 0.17689374089241028, "learning_rate": 0.00016659152250116812, "loss": 0.1029, "step": 175 }, { "epoch": 1.0, "grad_norm": 0.21386151015758514, "learning_rate": 0.00015594912061278626, "loss": 0.0918, "step": 200 }, { "epoch": 1.0, "eval_loss": 0.09140199422836304, "eval_runtime": 58.0967, "eval_samples_per_second": 3.443, "eval_steps_per_second": 0.861, "step": 200 }, { "epoch": 1.125, "grad_norm": 0.22129373252391815, "learning_rate": 0.00014428937460242417, "loss": 0.0694, "step": 225 }, { "epoch": 1.25, "grad_norm": 0.169610857963562, "learning_rate": 0.0001318242980627444, "loss": 0.0652, "step": 250 }, { "epoch": 1.375, "grad_norm": 0.2244400978088379, "learning_rate": 0.00011878054821746703, "loss": 0.0648, "step": 275 }, { "epoch": 1.5, "grad_norm": 0.15404100716114044, "learning_rate": 0.00010539530452693625, "loss": 0.0647, "step": 300 }, { "epoch": 1.625, "grad_norm": 0.1526843011379242, "learning_rate": 9.19119559638596e-05, "loss": 0.0549, "step": 325 }, { "epoch": 1.75, "grad_norm": 0.17617414891719818, "learning_rate": 7.857567537912403e-05, "loss": 0.0585, "step": 350 }, { "epoch": 1.875, "grad_norm": 0.18532665073871613, "learning_rate": 6.562896143067734e-05, "loss": 0.0606, "step": 375 }, { "epoch": 2.0, "grad_norm": 0.13493256270885468, "learning_rate": 5.3307229138275936e-05, "loss": 0.0531, "step": 400 }, { "epoch": 2.0, "eval_loss": 0.08346135914325714, "eval_runtime": 58.1824, "eval_samples_per_second": 3.437, "eval_steps_per_second": 0.859, "step": 400 }, { "epoch": 2.125, "grad_norm": 0.23295405507087708, "learning_rate": 4.183452924271776e-05, "loss": 0.031, "step": 425 }, { "epoch": 2.25, "grad_norm": 0.2066780924797058, "learning_rate": 3.1419474206078205e-05, "loss": 0.0319, "step": 450 }, { "epoch": 2.375, "grad_norm": 0.1161380186676979, "learning_rate": 2.2251444932035094e-05, "loss": 0.0306, "step": 475 }, { "epoch": 2.5, "grad_norm": 0.2266598790884018, "learning_rate": 1.4497147180928027e-05, "loss": 0.0259, "step": 500 }, { "epoch": 2.625, "grad_norm": 0.1240050196647644, "learning_rate": 8.297580295566575e-06, "loss": 0.0274, "step": 525 }, { "epoch": 2.75, "grad_norm": 0.13692007958889008, "learning_rate": 3.7654733565969826e-06, "loss": 0.0255, "step": 550 }, { "epoch": 2.875, "grad_norm": 0.1393318921327591, "learning_rate": 9.832353867893386e-07, "loss": 0.0269, "step": 575 }, { "epoch": 3.0, "grad_norm": 0.24243566393852234, "learning_rate": 1.4568764593603234e-09, "loss": 0.03, "step": 600 }, { "epoch": 3.0, "eval_loss": 0.10383956134319305, "eval_runtime": 58.2331, "eval_samples_per_second": 3.434, "eval_steps_per_second": 0.859, "step": 600 } ], "logging_steps": 25, "max_steps": 600, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 4.488064275802522e+16, "train_batch_size": 4, "trial_name": null, "trial_params": null }