devoppro commited on
Commit
40b81eb
·
verified ·
1 Parent(s): 0355b89

Training in progress, step 1100, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:135023d735474a62fef41f9fc86b17aa3dafa835f40463325b62635ce08a2f8b
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0b0e9b3f4ea4bfa18d8649a6588e6f31e5c42ffd86b85f74f3759d40af7fe5ec
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f89845f1e9c425ff77a21ebf1e6ff7f2fe52cc5d8d1eee15fffa7085d08ac1e6
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f83cff620522aa496468277a3bc22911f9c6f4bbc5118c2de2593b7cadedc59
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9aa7e5f7f3366b7db6f25a9e3fa739116674bd641ce589a5940ff73f382fec0c
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:35f81b1683bb9d9f30a4338a9869629a080b447f19358d02a2b979f0e40cf033
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4835dfebca207db20e97bc75a0cdfb3c9040e987a06d53d84bc51055c8c81f56
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff8a23e76bdb5af3bd48e8027611be72cb7ebf75d37e5954461baee2e16e1294
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.02,
6
  "eval_steps": 500,
7
- "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -708,6 +708,76 @@
708
  "learning_rate": 0.0002945951903807615,
709
  "loss": 71.14970703125,
710
  "step": 1000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
711
  }
712
  ],
713
  "logging_steps": 10,
@@ -727,7 +797,7 @@
727
  "attributes": {}
728
  }
729
  },
730
- "total_flos": 1.1337576305472e+16,
731
  "train_batch_size": 2,
732
  "trial_name": null,
733
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.022,
6
  "eval_steps": 500,
7
+ "global_step": 1100,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
708
  "learning_rate": 0.0002945951903807615,
709
  "loss": 71.14970703125,
710
  "step": 1000
711
+ },
712
+ {
713
+ "epoch": 0.0202,
714
+ "grad_norm": 9.4298677444458,
715
+ "learning_rate": 0.00029453507014028053,
716
+ "loss": 68.8579345703125,
717
+ "step": 1010
718
+ },
719
+ {
720
+ "epoch": 0.0204,
721
+ "grad_norm": 11.69355583190918,
722
+ "learning_rate": 0.0002944749498997996,
723
+ "loss": 66.29926147460938,
724
+ "step": 1020
725
+ },
726
+ {
727
+ "epoch": 0.0206,
728
+ "grad_norm": 9.638633728027344,
729
+ "learning_rate": 0.00029441482965931857,
730
+ "loss": 69.4812744140625,
731
+ "step": 1030
732
+ },
733
+ {
734
+ "epoch": 0.0208,
735
+ "grad_norm": 7.3949384689331055,
736
+ "learning_rate": 0.0002943547094188377,
737
+ "loss": 69.908984375,
738
+ "step": 1040
739
+ },
740
+ {
741
+ "epoch": 0.021,
742
+ "grad_norm": 9.138998031616211,
743
+ "learning_rate": 0.0002942945891783567,
744
+ "loss": 68.20003662109374,
745
+ "step": 1050
746
+ },
747
+ {
748
+ "epoch": 0.0212,
749
+ "grad_norm": 9.650430679321289,
750
+ "learning_rate": 0.0002942344689378757,
751
+ "loss": 64.76406860351562,
752
+ "step": 1060
753
+ },
754
+ {
755
+ "epoch": 0.0214,
756
+ "grad_norm": 8.263360977172852,
757
+ "learning_rate": 0.00029417434869739476,
758
+ "loss": 73.39156494140624,
759
+ "step": 1070
760
+ },
761
+ {
762
+ "epoch": 0.0216,
763
+ "grad_norm": 8.509857177734375,
764
+ "learning_rate": 0.0002941142284569138,
765
+ "loss": 65.55498657226562,
766
+ "step": 1080
767
+ },
768
+ {
769
+ "epoch": 0.0218,
770
+ "grad_norm": 10.162703514099121,
771
+ "learning_rate": 0.00029405410821643286,
772
+ "loss": 71.94191284179688,
773
+ "step": 1090
774
+ },
775
+ {
776
+ "epoch": 0.022,
777
+ "grad_norm": 10.938014030456543,
778
+ "learning_rate": 0.00029399398797595185,
779
+ "loss": 68.12153930664063,
780
+ "step": 1100
781
  }
782
  ],
783
  "logging_steps": 10,
 
797
  "attributes": {}
798
  }
799
  },
800
+ "total_flos": 1.19190649908096e+16,
801
  "train_batch_size": 2,
802
  "trial_name": null,
803
  "trial_params": null