devoppro commited on
Commit
dbd31a9
·
verified ·
1 Parent(s): 354aa36

Training in progress, step 1200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0b0e9b3f4ea4bfa18d8649a6588e6f31e5c42ffd86b85f74f3759d40af7fe5ec
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68d10d6c77661a22b49f687a60f7c0595cbbc5bdafd8820a654f5e22542b0e33
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f83cff620522aa496468277a3bc22911f9c6f4bbc5118c2de2593b7cadedc59
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:34eb8fd94484f12828f3079ddb1062719a88baf43040ad9bfe0255470c6e514a
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:35f81b1683bb9d9f30a4338a9869629a080b447f19358d02a2b979f0e40cf033
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:526edfc17cc77c8e1a37fdb81a9a703de4513b58ba4b0dbf11906a54aab78396
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ff8a23e76bdb5af3bd48e8027611be72cb7ebf75d37e5954461baee2e16e1294
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c692682bd43a00387bd41dcd12b266427cdc8a4657c91543547ea08ad6b10378
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.022,
6
  "eval_steps": 500,
7
- "global_step": 1100,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -778,6 +778,76 @@
778
  "learning_rate": 0.00029399398797595185,
779
  "loss": 68.12153930664063,
780
  "step": 1100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
781
  }
782
  ],
783
  "logging_steps": 10,
@@ -797,7 +867,7 @@
797
  "attributes": {}
798
  }
799
  },
800
- "total_flos": 1.19190649908096e+16,
801
  "train_batch_size": 2,
802
  "trial_name": null,
803
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.024,
6
  "eval_steps": 500,
7
+ "global_step": 1200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
778
  "learning_rate": 0.00029399398797595185,
779
  "loss": 68.12153930664063,
780
  "step": 1100
781
+ },
782
+ {
783
+ "epoch": 0.0222,
784
+ "grad_norm": 9.84087085723877,
785
+ "learning_rate": 0.0002939338677354709,
786
+ "loss": 73.03311767578126,
787
+ "step": 1110
788
+ },
789
+ {
790
+ "epoch": 0.0224,
791
+ "grad_norm": 8.5576171875,
792
+ "learning_rate": 0.00029387374749498995,
793
+ "loss": 67.728662109375,
794
+ "step": 1120
795
+ },
796
+ {
797
+ "epoch": 0.0226,
798
+ "grad_norm": 6.275845050811768,
799
+ "learning_rate": 0.000293813627254509,
800
+ "loss": 59.14338989257813,
801
+ "step": 1130
802
+ },
803
+ {
804
+ "epoch": 0.0228,
805
+ "grad_norm": 10.62365436553955,
806
+ "learning_rate": 0.00029375350701402804,
807
+ "loss": 68.015625,
808
+ "step": 1140
809
+ },
810
+ {
811
+ "epoch": 0.023,
812
+ "grad_norm": 10.124994277954102,
813
+ "learning_rate": 0.0002936933867735471,
814
+ "loss": 67.996630859375,
815
+ "step": 1150
816
+ },
817
+ {
818
+ "epoch": 0.0232,
819
+ "grad_norm": 9.192570686340332,
820
+ "learning_rate": 0.00029363326653306614,
821
+ "loss": 69.9107177734375,
822
+ "step": 1160
823
+ },
824
+ {
825
+ "epoch": 0.0234,
826
+ "grad_norm": 11.012198448181152,
827
+ "learning_rate": 0.00029357314629258513,
828
+ "loss": 72.0302978515625,
829
+ "step": 1170
830
+ },
831
+ {
832
+ "epoch": 0.0236,
833
+ "grad_norm": 8.58188533782959,
834
+ "learning_rate": 0.0002935130260521042,
835
+ "loss": 73.009423828125,
836
+ "step": 1180
837
+ },
838
+ {
839
+ "epoch": 0.0238,
840
+ "grad_norm": 11.214829444885254,
841
+ "learning_rate": 0.0002934529058116232,
842
+ "loss": 69.40020141601562,
843
+ "step": 1190
844
+ },
845
+ {
846
+ "epoch": 0.024,
847
+ "grad_norm": 8.615409851074219,
848
+ "learning_rate": 0.0002933927855711423,
849
+ "loss": 73.381103515625,
850
+ "step": 1200
851
  }
852
  ],
853
  "logging_steps": 10,
 
867
  "attributes": {}
868
  }
869
  },
870
+ "total_flos": 1.253963749312512e+16,
871
  "train_batch_size": 2,
872
  "trial_name": null,
873
  "trial_params": null