CodeIsAbstract commited on
Commit
dd7b5be
·
verified ·
1 Parent(s): b7ca72d

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:58d40820fb221ed5ab5ca4d9701da2049c6d1761cfc543431f711239f4c0a14f
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc4995194ae24360d283af7c87f66b06313c421ad3237dea5a676ea17582d264
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5169ba6435acf0ba6f9c584893daa0bc9782c7d8adc4d46b177c96492031e90b
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9134638ae868d7f042875926428ccfe72229e1aaba471b09bd4a42524bf529c7
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:363c5df1543d2c82b2f13164f35bdd0367ceb32e7fa1b2f67c19df073a08b17b
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:236304ae89e49aae8260113165ee63419b9b745f79120014328a7fa31ed79b42
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4abbae167a9f0ed409523b108568bee9d935380e25b792884224d262a0c2c795
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da9eb59c8b73626afc5a8c951a30627f32f26248a9fe83d71a280178d617963d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.1038961038961039,
6
  "eval_steps": 1000,
7
- "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -632,6 +632,318 @@
632
  "eval_samples_per_second": 45.514,
633
  "eval_steps_per_second": 11.378,
634
  "step": 8000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
635
  }
636
  ],
637
  "logging_steps": 100,
@@ -651,7 +963,7 @@
651
  "attributes": {}
652
  }
653
  },
654
- "total_flos": 2.3815451049984e+17,
655
  "train_batch_size": 22,
656
  "trial_name": null,
657
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.15584415584415584,
6
  "eval_steps": 1000,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
632
  "eval_samples_per_second": 45.514,
633
  "eval_steps_per_second": 11.378,
634
  "step": 8000
635
+ },
636
+ {
637
+ "epoch": 0.10519480519480519,
638
+ "grad_norm": 0.19026321172714233,
639
+ "learning_rate": 0.000989706938281085,
640
+ "loss": 3.0792,
641
+ "step": 8100
642
+ },
643
+ {
644
+ "epoch": 0.10649350649350649,
645
+ "grad_norm": 0.22636525332927704,
646
+ "learning_rate": 0.0009894138657894054,
647
+ "loss": 3.0808,
648
+ "step": 8200
649
+ },
650
+ {
651
+ "epoch": 0.10779220779220779,
652
+ "grad_norm": 0.1759577989578247,
653
+ "learning_rate": 0.0009891167239833311,
654
+ "loss": 3.0651,
655
+ "step": 8300
656
+ },
657
+ {
658
+ "epoch": 0.10909090909090909,
659
+ "grad_norm": 0.19504410028457642,
660
+ "learning_rate": 0.0009888155153334984,
661
+ "loss": 3.0701,
662
+ "step": 8400
663
+ },
664
+ {
665
+ "epoch": 0.11038961038961038,
666
+ "grad_norm": 0.20704291760921478,
667
+ "learning_rate": 0.000988510242344357,
668
+ "loss": 3.072,
669
+ "step": 8500
670
+ },
671
+ {
672
+ "epoch": 0.11168831168831168,
673
+ "grad_norm": 0.19917143881320953,
674
+ "learning_rate": 0.000988200907554151,
675
+ "loss": 3.0716,
676
+ "step": 8600
677
+ },
678
+ {
679
+ "epoch": 0.11298701298701298,
680
+ "grad_norm": 0.1709250509738922,
681
+ "learning_rate": 0.000987887513534897,
682
+ "loss": 3.0622,
683
+ "step": 8700
684
+ },
685
+ {
686
+ "epoch": 0.11428571428571428,
687
+ "grad_norm": 0.2797200381755829,
688
+ "learning_rate": 0.0009875700628923622,
689
+ "loss": 3.0743,
690
+ "step": 8800
691
+ },
692
+ {
693
+ "epoch": 0.11558441558441558,
694
+ "grad_norm": 0.172173872590065,
695
+ "learning_rate": 0.000987248558266044,
696
+ "loss": 3.0676,
697
+ "step": 8900
698
+ },
699
+ {
700
+ "epoch": 0.11688311688311688,
701
+ "grad_norm": 0.1997879594564438,
702
+ "learning_rate": 0.000986923002329147,
703
+ "loss": 3.0591,
704
+ "step": 9000
705
+ },
706
+ {
707
+ "epoch": 0.11688311688311688,
708
+ "eval_loss": 3.424792766571045,
709
+ "eval_runtime": 15.3808,
710
+ "eval_samples_per_second": 37.449,
711
+ "eval_steps_per_second": 9.362,
712
+ "step": 9000
713
+ },
714
+ {
715
+ "epoch": 0.11818181818181818,
716
+ "grad_norm": 0.16804350912570953,
717
+ "learning_rate": 0.0009865933977885612,
718
+ "loss": 3.0642,
719
+ "step": 9100
720
+ },
721
+ {
722
+ "epoch": 0.11948051948051948,
723
+ "grad_norm": 0.1866862177848816,
724
+ "learning_rate": 0.0009862597473848393,
725
+ "loss": 3.0378,
726
+ "step": 9200
727
+ },
728
+ {
729
+ "epoch": 0.12077922077922078,
730
+ "grad_norm": 0.2142857313156128,
731
+ "learning_rate": 0.000985922053892174,
732
+ "loss": 3.057,
733
+ "step": 9300
734
+ },
735
+ {
736
+ "epoch": 0.12207792207792208,
737
+ "grad_norm": 0.1772162914276123,
738
+ "learning_rate": 0.0009855803201183743,
739
+ "loss": 3.0585,
740
+ "step": 9400
741
+ },
742
+ {
743
+ "epoch": 0.12337662337662338,
744
+ "grad_norm": 0.1796869933605194,
745
+ "learning_rate": 0.0009852345489048447,
746
+ "loss": 3.0228,
747
+ "step": 9500
748
+ },
749
+ {
750
+ "epoch": 0.12467532467532468,
751
+ "grad_norm": 0.19519853591918945,
752
+ "learning_rate": 0.0009848847431265576,
753
+ "loss": 3.0493,
754
+ "step": 9600
755
+ },
756
+ {
757
+ "epoch": 0.12597402597402596,
758
+ "grad_norm": 0.1866455227136612,
759
+ "learning_rate": 0.0009845309056920326,
760
+ "loss": 3.0369,
761
+ "step": 9700
762
+ },
763
+ {
764
+ "epoch": 0.12727272727272726,
765
+ "grad_norm": 0.1888640969991684,
766
+ "learning_rate": 0.000984173039543311,
767
+ "loss": 3.0388,
768
+ "step": 9800
769
+ },
770
+ {
771
+ "epoch": 0.12857142857142856,
772
+ "grad_norm": 0.1753017008304596,
773
+ "learning_rate": 0.0009838111476559313,
774
+ "loss": 3.0533,
775
+ "step": 9900
776
+ },
777
+ {
778
+ "epoch": 0.12987012987012986,
779
+ "grad_norm": 0.17457301914691925,
780
+ "learning_rate": 0.000983445233038905,
781
+ "loss": 3.0569,
782
+ "step": 10000
783
+ },
784
+ {
785
+ "epoch": 0.12987012987012986,
786
+ "eval_loss": 3.401273012161255,
787
+ "eval_runtime": 14.6757,
788
+ "eval_samples_per_second": 39.248,
789
+ "eval_steps_per_second": 9.812,
790
+ "step": 10000
791
+ },
792
+ {
793
+ "epoch": 0.13116883116883116,
794
+ "grad_norm": 0.17421157658100128,
795
+ "learning_rate": 0.0009830752987346908,
796
+ "loss": 3.0596,
797
+ "step": 10100
798
+ },
799
+ {
800
+ "epoch": 0.13246753246753246,
801
+ "grad_norm": 0.1772899180650711,
802
+ "learning_rate": 0.0009827013478191703,
803
+ "loss": 3.016,
804
+ "step": 10200
805
+ },
806
+ {
807
+ "epoch": 0.13376623376623376,
808
+ "grad_norm": 0.169020414352417,
809
+ "learning_rate": 0.0009823233834016214,
810
+ "loss": 3.0173,
811
+ "step": 10300
812
+ },
813
+ {
814
+ "epoch": 0.13506493506493505,
815
+ "grad_norm": 0.34023940563201904,
816
+ "learning_rate": 0.0009819414086246938,
817
+ "loss": 3.0004,
818
+ "step": 10400
819
+ },
820
+ {
821
+ "epoch": 0.13636363636363635,
822
+ "grad_norm": 0.19924494624137878,
823
+ "learning_rate": 0.0009815554266643808,
824
+ "loss": 3.003,
825
+ "step": 10500
826
+ },
827
+ {
828
+ "epoch": 0.13766233766233765,
829
+ "grad_norm": 0.1713639199733734,
830
+ "learning_rate": 0.0009811654407299948,
831
+ "loss": 2.9895,
832
+ "step": 10600
833
+ },
834
+ {
835
+ "epoch": 0.13896103896103895,
836
+ "grad_norm": 0.16809602081775665,
837
+ "learning_rate": 0.00098077145406414,
838
+ "loss": 3.0078,
839
+ "step": 10700
840
+ },
841
+ {
842
+ "epoch": 0.14025974025974025,
843
+ "grad_norm": 0.18249379098415375,
844
+ "learning_rate": 0.0009803734699426853,
845
+ "loss": 3.0379,
846
+ "step": 10800
847
+ },
848
+ {
849
+ "epoch": 0.14155844155844155,
850
+ "grad_norm": 0.19280089437961578,
851
+ "learning_rate": 0.0009799714916747368,
852
+ "loss": 2.9917,
853
+ "step": 10900
854
+ },
855
+ {
856
+ "epoch": 0.14285714285714285,
857
+ "grad_norm": 0.18399550020694733,
858
+ "learning_rate": 0.000979565522602611,
859
+ "loss": 3.0597,
860
+ "step": 11000
861
+ },
862
+ {
863
+ "epoch": 0.14285714285714285,
864
+ "eval_loss": 3.4125261306762695,
865
+ "eval_runtime": 15.5875,
866
+ "eval_samples_per_second": 36.953,
867
+ "eval_steps_per_second": 9.238,
868
+ "step": 11000
869
+ },
870
+ {
871
+ "epoch": 0.14415584415584415,
872
+ "grad_norm": 0.17574363946914673,
873
+ "learning_rate": 0.000979155566101806,
874
+ "loss": 3.0168,
875
+ "step": 11100
876
+ },
877
+ {
878
+ "epoch": 0.14545454545454545,
879
+ "grad_norm": 0.17517109215259552,
880
+ "learning_rate": 0.0009787416255809752,
881
+ "loss": 2.9817,
882
+ "step": 11200
883
+ },
884
+ {
885
+ "epoch": 0.14675324675324675,
886
+ "grad_norm": 0.18359977006912231,
887
+ "learning_rate": 0.0009783237044818968,
888
+ "loss": 2.9916,
889
+ "step": 11300
890
+ },
891
+ {
892
+ "epoch": 0.14805194805194805,
893
+ "grad_norm": 0.17518875002861023,
894
+ "learning_rate": 0.000977901806279446,
895
+ "loss": 2.9567,
896
+ "step": 11400
897
+ },
898
+ {
899
+ "epoch": 0.14935064935064934,
900
+ "grad_norm": 0.19247221946716309,
901
+ "learning_rate": 0.0009774759344815674,
902
+ "loss": 2.9901,
903
+ "step": 11500
904
+ },
905
+ {
906
+ "epoch": 0.15064935064935064,
907
+ "grad_norm": 0.18513117730617523,
908
+ "learning_rate": 0.000977046092629244,
909
+ "loss": 2.992,
910
+ "step": 11600
911
+ },
912
+ {
913
+ "epoch": 0.15194805194805194,
914
+ "grad_norm": 0.18412470817565918,
915
+ "learning_rate": 0.0009766122842964683,
916
+ "loss": 2.9985,
917
+ "step": 11700
918
+ },
919
+ {
920
+ "epoch": 0.15324675324675324,
921
+ "grad_norm": 0.19853200018405914,
922
+ "learning_rate": 0.0009761745130902134,
923
+ "loss": 2.978,
924
+ "step": 11800
925
+ },
926
+ {
927
+ "epoch": 0.15454545454545454,
928
+ "grad_norm": 0.19611628353595734,
929
+ "learning_rate": 0.0009757327826504022,
930
+ "loss": 2.9771,
931
+ "step": 11900
932
+ },
933
+ {
934
+ "epoch": 0.15584415584415584,
935
+ "grad_norm": 0.17522920668125153,
936
+ "learning_rate": 0.0009752870966498766,
937
+ "loss": 2.9641,
938
+ "step": 12000
939
+ },
940
+ {
941
+ "epoch": 0.15584415584415584,
942
+ "eval_loss": 3.3487231731414795,
943
+ "eval_runtime": 13.9045,
944
+ "eval_samples_per_second": 41.425,
945
+ "eval_steps_per_second": 10.356,
946
+ "step": 12000
947
  }
948
  ],
949
  "logging_steps": 100,
 
963
  "attributes": {}
964
  }
965
  },
966
+ "total_flos": 3.5723176574976e+17,
967
  "train_batch_size": 22,
968
  "trial_name": null,
969
  "trial_params": null