CodeIsAbstract commited on
Commit
7550bd4
·
verified ·
1 Parent(s): bafb0b2

Training in progress, step 1500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6474fca5a3c5baba3de6445b1ebdccd489d640243fc43690efa828d0c4fe3868
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:89952de3671811784fe7b357f32398d8262fdfa34181cc91ec5b8a5baba9ebc3
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e45d306cbed8140a90bd411a1fece5d9556ba1ea6e19977f26931c3c04004bd4
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0192b698deb59413cfa00577b1ee14f0ef3ba0e14eb3aa418fb3f772003b2fd3
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e1b8777afaf65f04eb18ddaf41f68fd24275d3e886bf75d506bd86a6d10df7f8
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8c2ed9303a8f4e0d19061969cd3c3a9b1a1fd302644d59f1e023ca72324b1d0
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dfc5c6949cf8a9eca4d88710f0c9c06f410fb63454fb60c644727889ad457648
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:793c5540f8117a4ddd149538354eaaa33f8c7f011c83883ddf29fe821afe64fa
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2,
6
  "eval_steps": 100,
7
- "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -788,6 +788,396 @@
788
  "eval_samples_per_second": 67.186,
789
  "eval_steps_per_second": 3.833,
790
  "step": 1000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
791
  }
792
  ],
793
  "logging_steps": 10,
@@ -807,7 +1197,7 @@
807
  "attributes": {}
808
  }
809
  },
810
- "total_flos": 3.26154514857984e+17,
811
  "train_batch_size": 18,
812
  "trial_name": null,
813
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.3,
6
  "eval_steps": 100,
7
+ "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
788
  "eval_samples_per_second": 67.186,
789
  "eval_steps_per_second": 3.833,
790
  "step": 1000
791
+ },
792
+ {
793
+ "epoch": 0.202,
794
+ "grad_norm": 0.0859375,
795
+ "learning_rate": 0.0003,
796
+ "loss": 2.621230697631836,
797
+ "step": 1010
798
+ },
799
+ {
800
+ "epoch": 0.204,
801
+ "grad_norm": 0.056640625,
802
+ "learning_rate": 0.0003,
803
+ "loss": 2.6316347122192383,
804
+ "step": 1020
805
+ },
806
+ {
807
+ "epoch": 0.206,
808
+ "grad_norm": 0.2431640625,
809
+ "learning_rate": 0.0003,
810
+ "loss": 2.6155324935913087,
811
+ "step": 1030
812
+ },
813
+ {
814
+ "epoch": 0.208,
815
+ "grad_norm": 0.06298828125,
816
+ "learning_rate": 0.0003,
817
+ "loss": 2.634638214111328,
818
+ "step": 1040
819
+ },
820
+ {
821
+ "epoch": 0.21,
822
+ "grad_norm": 0.1064453125,
823
+ "learning_rate": 0.0003,
824
+ "loss": 2.6131080627441405,
825
+ "step": 1050
826
+ },
827
+ {
828
+ "epoch": 0.212,
829
+ "grad_norm": 0.06689453125,
830
+ "learning_rate": 0.0003,
831
+ "loss": 2.616171646118164,
832
+ "step": 1060
833
+ },
834
+ {
835
+ "epoch": 0.214,
836
+ "grad_norm": 0.08984375,
837
+ "learning_rate": 0.0003,
838
+ "loss": 2.603512191772461,
839
+ "step": 1070
840
+ },
841
+ {
842
+ "epoch": 0.216,
843
+ "grad_norm": 0.0625,
844
+ "learning_rate": 0.0003,
845
+ "loss": 2.6277271270751954,
846
+ "step": 1080
847
+ },
848
+ {
849
+ "epoch": 0.218,
850
+ "grad_norm": 0.0673828125,
851
+ "learning_rate": 0.0003,
852
+ "loss": 2.615810775756836,
853
+ "step": 1090
854
+ },
855
+ {
856
+ "epoch": 0.22,
857
+ "grad_norm": 0.060791015625,
858
+ "learning_rate": 0.0003,
859
+ "loss": 2.6153131484985352,
860
+ "step": 1100
861
+ },
862
+ {
863
+ "epoch": 0.22,
864
+ "eval_loss": 3.0302107334136963,
865
+ "eval_runtime": 4.4032,
866
+ "eval_samples_per_second": 67.678,
867
+ "eval_steps_per_second": 3.861,
868
+ "step": 1100
869
+ },
870
+ {
871
+ "epoch": 0.222,
872
+ "grad_norm": 0.064453125,
873
+ "learning_rate": 0.0003,
874
+ "loss": 2.623016357421875,
875
+ "step": 1110
876
+ },
877
+ {
878
+ "epoch": 0.224,
879
+ "grad_norm": 0.0712890625,
880
+ "learning_rate": 0.0003,
881
+ "loss": 2.6170467376708983,
882
+ "step": 1120
883
+ },
884
+ {
885
+ "epoch": 0.226,
886
+ "grad_norm": 0.59375,
887
+ "learning_rate": 0.0003,
888
+ "loss": 2.620796775817871,
889
+ "step": 1130
890
+ },
891
+ {
892
+ "epoch": 0.228,
893
+ "grad_norm": 0.30078125,
894
+ "learning_rate": 0.0003,
895
+ "loss": 2.6082935333251953,
896
+ "step": 1140
897
+ },
898
+ {
899
+ "epoch": 0.23,
900
+ "grad_norm": 0.80078125,
901
+ "learning_rate": 0.0003,
902
+ "loss": 2.6210090637207033,
903
+ "step": 1150
904
+ },
905
+ {
906
+ "epoch": 0.232,
907
+ "grad_norm": 0.0634765625,
908
+ "learning_rate": 0.0003,
909
+ "loss": 2.636818695068359,
910
+ "step": 1160
911
+ },
912
+ {
913
+ "epoch": 0.234,
914
+ "grad_norm": 0.07080078125,
915
+ "learning_rate": 0.0003,
916
+ "loss": 2.5969371795654297,
917
+ "step": 1170
918
+ },
919
+ {
920
+ "epoch": 0.236,
921
+ "grad_norm": 0.107421875,
922
+ "learning_rate": 0.0003,
923
+ "loss": 2.6085494995117187,
924
+ "step": 1180
925
+ },
926
+ {
927
+ "epoch": 0.238,
928
+ "grad_norm": 0.1826171875,
929
+ "learning_rate": 0.0003,
930
+ "loss": 2.6118932723999024,
931
+ "step": 1190
932
+ },
933
+ {
934
+ "epoch": 0.24,
935
+ "grad_norm": 0.0625,
936
+ "learning_rate": 0.0003,
937
+ "loss": 2.6277448654174806,
938
+ "step": 1200
939
+ },
940
+ {
941
+ "epoch": 0.24,
942
+ "eval_loss": 3.0262646675109863,
943
+ "eval_runtime": 4.3768,
944
+ "eval_samples_per_second": 68.087,
945
+ "eval_steps_per_second": 3.884,
946
+ "step": 1200
947
+ },
948
+ {
949
+ "epoch": 0.242,
950
+ "grad_norm": 0.08837890625,
951
+ "learning_rate": 0.0003,
952
+ "loss": 2.6076292037963866,
953
+ "step": 1210
954
+ },
955
+ {
956
+ "epoch": 0.244,
957
+ "grad_norm": 0.0869140625,
958
+ "learning_rate": 0.0003,
959
+ "loss": 2.6239038467407227,
960
+ "step": 1220
961
+ },
962
+ {
963
+ "epoch": 0.246,
964
+ "grad_norm": 0.06201171875,
965
+ "learning_rate": 0.0003,
966
+ "loss": 2.598759651184082,
967
+ "step": 1230
968
+ },
969
+ {
970
+ "epoch": 0.248,
971
+ "grad_norm": 0.054443359375,
972
+ "learning_rate": 0.0003,
973
+ "loss": 2.623333549499512,
974
+ "step": 1240
975
+ },
976
+ {
977
+ "epoch": 0.25,
978
+ "grad_norm": 0.056884765625,
979
+ "learning_rate": 0.0003,
980
+ "loss": 2.595660400390625,
981
+ "step": 1250
982
+ },
983
+ {
984
+ "epoch": 0.252,
985
+ "grad_norm": 0.07470703125,
986
+ "learning_rate": 0.0003,
987
+ "loss": 2.6085866928100585,
988
+ "step": 1260
989
+ },
990
+ {
991
+ "epoch": 0.254,
992
+ "grad_norm": 0.06201171875,
993
+ "learning_rate": 0.0003,
994
+ "loss": 2.6313194274902343,
995
+ "step": 1270
996
+ },
997
+ {
998
+ "epoch": 0.256,
999
+ "grad_norm": 0.05126953125,
1000
+ "learning_rate": 0.0003,
1001
+ "loss": 2.624466323852539,
1002
+ "step": 1280
1003
+ },
1004
+ {
1005
+ "epoch": 0.258,
1006
+ "grad_norm": 0.0693359375,
1007
+ "learning_rate": 0.0003,
1008
+ "loss": 2.61032657623291,
1009
+ "step": 1290
1010
+ },
1011
+ {
1012
+ "epoch": 0.26,
1013
+ "grad_norm": 0.95703125,
1014
+ "learning_rate": 0.0003,
1015
+ "loss": 2.611441230773926,
1016
+ "step": 1300
1017
+ },
1018
+ {
1019
+ "epoch": 0.26,
1020
+ "eval_loss": 3.0270800590515137,
1021
+ "eval_runtime": 4.4364,
1022
+ "eval_samples_per_second": 67.171,
1023
+ "eval_steps_per_second": 3.832,
1024
+ "step": 1300
1025
+ },
1026
+ {
1027
+ "epoch": 0.262,
1028
+ "grad_norm": 0.416015625,
1029
+ "learning_rate": 0.0003,
1030
+ "loss": 2.622879981994629,
1031
+ "step": 1310
1032
+ },
1033
+ {
1034
+ "epoch": 0.264,
1035
+ "grad_norm": 2.015625,
1036
+ "learning_rate": 0.0003,
1037
+ "loss": 2.6255945205688476,
1038
+ "step": 1320
1039
+ },
1040
+ {
1041
+ "epoch": 0.266,
1042
+ "grad_norm": 0.419921875,
1043
+ "learning_rate": 0.0003,
1044
+ "loss": 2.619590950012207,
1045
+ "step": 1330
1046
+ },
1047
+ {
1048
+ "epoch": 0.268,
1049
+ "grad_norm": 0.057861328125,
1050
+ "learning_rate": 0.0003,
1051
+ "loss": 2.618213081359863,
1052
+ "step": 1340
1053
+ },
1054
+ {
1055
+ "epoch": 0.27,
1056
+ "grad_norm": 0.054443359375,
1057
+ "learning_rate": 0.0003,
1058
+ "loss": 2.6290687561035155,
1059
+ "step": 1350
1060
+ },
1061
+ {
1062
+ "epoch": 0.272,
1063
+ "grad_norm": 0.2080078125,
1064
+ "learning_rate": 0.0003,
1065
+ "loss": 2.5912906646728517,
1066
+ "step": 1360
1067
+ },
1068
+ {
1069
+ "epoch": 0.274,
1070
+ "grad_norm": 0.06787109375,
1071
+ "learning_rate": 0.0003,
1072
+ "loss": 2.608586311340332,
1073
+ "step": 1370
1074
+ },
1075
+ {
1076
+ "epoch": 0.276,
1077
+ "grad_norm": 0.2734375,
1078
+ "learning_rate": 0.0003,
1079
+ "loss": 2.6140621185302733,
1080
+ "step": 1380
1081
+ },
1082
+ {
1083
+ "epoch": 0.278,
1084
+ "grad_norm": 0.060546875,
1085
+ "learning_rate": 0.0003,
1086
+ "loss": 2.6242074966430664,
1087
+ "step": 1390
1088
+ },
1089
+ {
1090
+ "epoch": 0.28,
1091
+ "grad_norm": 0.056640625,
1092
+ "learning_rate": 0.0003,
1093
+ "loss": 2.6025035858154295,
1094
+ "step": 1400
1095
+ },
1096
+ {
1097
+ "epoch": 0.28,
1098
+ "eval_loss": 3.0289111137390137,
1099
+ "eval_runtime": 4.3992,
1100
+ "eval_samples_per_second": 67.739,
1101
+ "eval_steps_per_second": 3.864,
1102
+ "step": 1400
1103
+ },
1104
+ {
1105
+ "epoch": 0.282,
1106
+ "grad_norm": 0.09228515625,
1107
+ "learning_rate": 0.0003,
1108
+ "loss": 2.6210390090942384,
1109
+ "step": 1410
1110
+ },
1111
+ {
1112
+ "epoch": 0.284,
1113
+ "grad_norm": 0.087890625,
1114
+ "learning_rate": 0.0003,
1115
+ "loss": 2.6299697875976564,
1116
+ "step": 1420
1117
+ },
1118
+ {
1119
+ "epoch": 0.286,
1120
+ "grad_norm": 0.060546875,
1121
+ "learning_rate": 0.0003,
1122
+ "loss": 2.60406494140625,
1123
+ "step": 1430
1124
+ },
1125
+ {
1126
+ "epoch": 0.288,
1127
+ "grad_norm": 0.060546875,
1128
+ "learning_rate": 0.0003,
1129
+ "loss": 2.599949264526367,
1130
+ "step": 1440
1131
+ },
1132
+ {
1133
+ "epoch": 0.29,
1134
+ "grad_norm": 0.212890625,
1135
+ "learning_rate": 0.0003,
1136
+ "loss": 2.5918996810913084,
1137
+ "step": 1450
1138
+ },
1139
+ {
1140
+ "epoch": 0.292,
1141
+ "grad_norm": 0.51953125,
1142
+ "learning_rate": 0.0003,
1143
+ "loss": 2.5961977005004884,
1144
+ "step": 1460
1145
+ },
1146
+ {
1147
+ "epoch": 0.294,
1148
+ "grad_norm": 0.0634765625,
1149
+ "learning_rate": 0.0003,
1150
+ "loss": 2.5982738494873048,
1151
+ "step": 1470
1152
+ },
1153
+ {
1154
+ "epoch": 0.296,
1155
+ "grad_norm": 0.0771484375,
1156
+ "learning_rate": 0.0003,
1157
+ "loss": 2.6246980667114257,
1158
+ "step": 1480
1159
+ },
1160
+ {
1161
+ "epoch": 0.298,
1162
+ "grad_norm": 0.515625,
1163
+ "learning_rate": 0.0003,
1164
+ "loss": 2.583383560180664,
1165
+ "step": 1490
1166
+ },
1167
+ {
1168
+ "epoch": 0.3,
1169
+ "grad_norm": 0.068359375,
1170
+ "learning_rate": 0.0003,
1171
+ "loss": 2.6035650253295897,
1172
+ "step": 1500
1173
+ },
1174
+ {
1175
+ "epoch": 0.3,
1176
+ "eval_loss": 3.0258114337921143,
1177
+ "eval_runtime": 4.3825,
1178
+ "eval_samples_per_second": 67.998,
1179
+ "eval_steps_per_second": 3.879,
1180
+ "step": 1500
1181
  }
1182
  ],
1183
  "logging_steps": 10,
 
1197
  "attributes": {}
1198
  }
1199
  },
1200
+ "total_flos": 4.89231772286976e+17,
1201
  "train_batch_size": 18,
1202
  "trial_name": null,
1203
  "trial_params": null