CodeIsAbstract commited on
Commit
86490d9
·
verified ·
1 Parent(s): e33bc4e

Training in progress, step 16000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fc4995194ae24360d283af7c87f66b06313c421ad3237dea5a676ea17582d264
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eee590a6a542d1f1d5b11450eee8ea5e55218ea6dd1f5ebb31fcfdd5d6173a1d
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9134638ae868d7f042875926428ccfe72229e1aaba471b09bd4a42524bf529c7
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:169d12341cc26c42de660d44a72be6e4103b1e6138612f4b5da9fb2664cd037d
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:236304ae89e49aae8260113165ee63419b9b745f79120014328a7fa31ed79b42
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b928d2d8033ac6bd87c58a39b741faace8bd1c6b0d070b7fad23c19520ff9f1a
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:da9eb59c8b73626afc5a8c951a30627f32f26248a9fe83d71a280178d617963d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ea5f9e2ce496bc8a089f39581e874b6a1c3a092591c67b5a4c3de15a05a65b33
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.15584415584415584,
6
  "eval_steps": 1000,
7
- "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -944,6 +944,318 @@
944
  "eval_samples_per_second": 41.425,
945
  "eval_steps_per_second": 10.356,
946
  "step": 12000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
947
  }
948
  ],
949
  "logging_steps": 100,
@@ -963,7 +1275,7 @@
963
  "attributes": {}
964
  }
965
  },
966
- "total_flos": 3.5723176574976e+17,
967
  "train_batch_size": 22,
968
  "trial_name": null,
969
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2077922077922078,
6
  "eval_steps": 1000,
7
+ "global_step": 16000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
944
  "eval_samples_per_second": 41.425,
945
  "eval_steps_per_second": 10.356,
946
  "step": 12000
947
+ },
948
+ {
949
+ "epoch": 0.15714285714285714,
950
+ "grad_norm": 0.18391777575016022,
951
+ "learning_rate": 0.0009748374587943688,
952
+ "loss": 2.9694,
953
+ "step": 12100
954
+ },
955
+ {
956
+ "epoch": 0.15844155844155844,
957
+ "grad_norm": 0.20561572909355164,
958
+ "learning_rate": 0.0009743838728224687,
959
+ "loss": 3.0187,
960
+ "step": 12200
961
+ },
962
+ {
963
+ "epoch": 0.15974025974025974,
964
+ "grad_norm": 0.17854483425617218,
965
+ "learning_rate": 0.0009739263425055934,
966
+ "loss": 2.9687,
967
+ "step": 12300
968
+ },
969
+ {
970
+ "epoch": 0.16103896103896104,
971
+ "grad_norm": 0.18647781014442444,
972
+ "learning_rate": 0.0009734648716479563,
973
+ "loss": 2.9574,
974
+ "step": 12400
975
+ },
976
+ {
977
+ "epoch": 0.16233766233766234,
978
+ "grad_norm": 0.1802452653646469,
979
+ "learning_rate": 0.0009729994640865349,
980
+ "loss": 2.9812,
981
+ "step": 12500
982
+ },
983
+ {
984
+ "epoch": 0.16363636363636364,
985
+ "grad_norm": 0.18787842988967896,
986
+ "learning_rate": 0.0009725301236910393,
987
+ "loss": 2.9771,
988
+ "step": 12600
989
+ },
990
+ {
991
+ "epoch": 0.16493506493506493,
992
+ "grad_norm": 0.18528838455677032,
993
+ "learning_rate": 0.0009720568543638793,
994
+ "loss": 2.9525,
995
+ "step": 12700
996
+ },
997
+ {
998
+ "epoch": 0.16623376623376623,
999
+ "grad_norm": 0.17883966863155365,
1000
+ "learning_rate": 0.000971579660040133,
1001
+ "loss": 2.9624,
1002
+ "step": 12800
1003
+ },
1004
+ {
1005
+ "epoch": 0.16753246753246753,
1006
+ "grad_norm": 0.17459270358085632,
1007
+ "learning_rate": 0.0009710985446875134,
1008
+ "loss": 2.9696,
1009
+ "step": 12900
1010
+ },
1011
+ {
1012
+ "epoch": 0.16883116883116883,
1013
+ "grad_norm": 0.1830231249332428,
1014
+ "learning_rate": 0.0009706135123063352,
1015
+ "loss": 2.9609,
1016
+ "step": 13000
1017
+ },
1018
+ {
1019
+ "epoch": 0.16883116883116883,
1020
+ "eval_loss": 3.337019920349121,
1021
+ "eval_runtime": 15.8318,
1022
+ "eval_samples_per_second": 36.383,
1023
+ "eval_steps_per_second": 9.096,
1024
+ "step": 13000
1025
+ },
1026
+ {
1027
+ "epoch": 0.17012987012987013,
1028
+ "grad_norm": 0.20791789889335632,
1029
+ "learning_rate": 0.0009701245669294825,
1030
+ "loss": 2.9953,
1031
+ "step": 13100
1032
+ },
1033
+ {
1034
+ "epoch": 0.17142857142857143,
1035
+ "grad_norm": 0.20302486419677734,
1036
+ "learning_rate": 0.0009696317126223742,
1037
+ "loss": 3.0519,
1038
+ "step": 13200
1039
+ },
1040
+ {
1041
+ "epoch": 0.17272727272727273,
1042
+ "grad_norm": 0.17939983308315277,
1043
+ "learning_rate": 0.000969134953482931,
1044
+ "loss": 2.9618,
1045
+ "step": 13300
1046
+ },
1047
+ {
1048
+ "epoch": 0.17402597402597403,
1049
+ "grad_norm": 0.2437734603881836,
1050
+ "learning_rate": 0.0009686342936415407,
1051
+ "loss": 2.955,
1052
+ "step": 13400
1053
+ },
1054
+ {
1055
+ "epoch": 0.17532467532467533,
1056
+ "grad_norm": 0.19909310340881348,
1057
+ "learning_rate": 0.000968129737261024,
1058
+ "loss": 2.9898,
1059
+ "step": 13500
1060
+ },
1061
+ {
1062
+ "epoch": 0.17662337662337663,
1063
+ "grad_norm": 0.1696212738752365,
1064
+ "learning_rate": 0.000967621288536601,
1065
+ "loss": 2.9793,
1066
+ "step": 13600
1067
+ },
1068
+ {
1069
+ "epoch": 0.17792207792207793,
1070
+ "grad_norm": 0.171333447098732,
1071
+ "learning_rate": 0.0009671089516958538,
1072
+ "loss": 2.942,
1073
+ "step": 13700
1074
+ },
1075
+ {
1076
+ "epoch": 0.17922077922077922,
1077
+ "grad_norm": 0.17436903715133667,
1078
+ "learning_rate": 0.0009665927309986944,
1079
+ "loss": 2.9574,
1080
+ "step": 13800
1081
+ },
1082
+ {
1083
+ "epoch": 0.18051948051948052,
1084
+ "grad_norm": 0.20607399940490723,
1085
+ "learning_rate": 0.0009660726307373266,
1086
+ "loss": 2.965,
1087
+ "step": 13900
1088
+ },
1089
+ {
1090
+ "epoch": 0.18181818181818182,
1091
+ "grad_norm": 0.17206591367721558,
1092
+ "learning_rate": 0.0009655486552362127,
1093
+ "loss": 2.9532,
1094
+ "step": 14000
1095
+ },
1096
+ {
1097
+ "epoch": 0.18181818181818182,
1098
+ "eval_loss": 3.327735424041748,
1099
+ "eval_runtime": 13.5364,
1100
+ "eval_samples_per_second": 42.552,
1101
+ "eval_steps_per_second": 10.638,
1102
+ "step": 14000
1103
+ },
1104
+ {
1105
+ "epoch": 0.18311688311688312,
1106
+ "grad_norm": 0.20287004113197327,
1107
+ "learning_rate": 0.000965020808852035,
1108
+ "loss": 2.9153,
1109
+ "step": 14100
1110
+ },
1111
+ {
1112
+ "epoch": 0.18441558441558442,
1113
+ "grad_norm": 0.18036919832229614,
1114
+ "learning_rate": 0.0009644890959736619,
1115
+ "loss": 2.933,
1116
+ "step": 14200
1117
+ },
1118
+ {
1119
+ "epoch": 0.18571428571428572,
1120
+ "grad_norm": 0.21330875158309937,
1121
+ "learning_rate": 0.0009639535210221099,
1122
+ "loss": 2.9773,
1123
+ "step": 14300
1124
+ },
1125
+ {
1126
+ "epoch": 0.18701298701298702,
1127
+ "grad_norm": 0.20942148566246033,
1128
+ "learning_rate": 0.000963414088450508,
1129
+ "loss": 2.9225,
1130
+ "step": 14400
1131
+ },
1132
+ {
1133
+ "epoch": 0.18831168831168832,
1134
+ "grad_norm": 0.188313826918602,
1135
+ "learning_rate": 0.0009628708027440592,
1136
+ "loss": 2.9409,
1137
+ "step": 14500
1138
+ },
1139
+ {
1140
+ "epoch": 0.18961038961038962,
1141
+ "grad_norm": 0.17884506285190582,
1142
+ "learning_rate": 0.0009623236684200043,
1143
+ "loss": 2.9635,
1144
+ "step": 14600
1145
+ },
1146
+ {
1147
+ "epoch": 0.19090909090909092,
1148
+ "grad_norm": 0.17130698263645172,
1149
+ "learning_rate": 0.0009617726900275845,
1150
+ "loss": 2.9447,
1151
+ "step": 14700
1152
+ },
1153
+ {
1154
+ "epoch": 0.19220779220779222,
1155
+ "grad_norm": 0.19943124055862427,
1156
+ "learning_rate": 0.0009612178721480027,
1157
+ "loss": 2.9525,
1158
+ "step": 14800
1159
+ },
1160
+ {
1161
+ "epoch": 0.19350649350649352,
1162
+ "grad_norm": 0.18249951303005219,
1163
+ "learning_rate": 0.0009606592193943861,
1164
+ "loss": 2.9405,
1165
+ "step": 14900
1166
+ },
1167
+ {
1168
+ "epoch": 0.19480519480519481,
1169
+ "grad_norm": 0.17710338532924652,
1170
+ "learning_rate": 0.0009600967364117478,
1171
+ "loss": 2.921,
1172
+ "step": 15000
1173
+ },
1174
+ {
1175
+ "epoch": 0.19480519480519481,
1176
+ "eval_loss": 3.307467460632324,
1177
+ "eval_runtime": 15.2964,
1178
+ "eval_samples_per_second": 37.656,
1179
+ "eval_steps_per_second": 9.414,
1180
+ "step": 15000
1181
+ },
1182
+ {
1183
+ "epoch": 0.1961038961038961,
1184
+ "grad_norm": 0.2033987194299698,
1185
+ "learning_rate": 0.0009595304278769472,
1186
+ "loss": 2.9253,
1187
+ "step": 15100
1188
+ },
1189
+ {
1190
+ "epoch": 0.1974025974025974,
1191
+ "grad_norm": 0.17457351088523865,
1192
+ "learning_rate": 0.000958960298498653,
1193
+ "loss": 2.9183,
1194
+ "step": 15200
1195
+ },
1196
+ {
1197
+ "epoch": 0.1987012987012987,
1198
+ "grad_norm": 0.2011554092168808,
1199
+ "learning_rate": 0.0009583863530173018,
1200
+ "loss": 2.9254,
1201
+ "step": 15300
1202
+ },
1203
+ {
1204
+ "epoch": 0.2,
1205
+ "grad_norm": 0.18333646655082703,
1206
+ "learning_rate": 0.0009578085962050609,
1207
+ "loss": 2.9383,
1208
+ "step": 15400
1209
+ },
1210
+ {
1211
+ "epoch": 0.2012987012987013,
1212
+ "grad_norm": 0.167351633310318,
1213
+ "learning_rate": 0.0009572270328657869,
1214
+ "loss": 2.9435,
1215
+ "step": 15500
1216
+ },
1217
+ {
1218
+ "epoch": 0.2025974025974026,
1219
+ "grad_norm": 0.17198020219802856,
1220
+ "learning_rate": 0.0009566416678349864,
1221
+ "loss": 2.9336,
1222
+ "step": 15600
1223
+ },
1224
+ {
1225
+ "epoch": 0.2038961038961039,
1226
+ "grad_norm": 0.19012530148029327,
1227
+ "learning_rate": 0.0009560525059797762,
1228
+ "loss": 2.9353,
1229
+ "step": 15700
1230
+ },
1231
+ {
1232
+ "epoch": 0.2051948051948052,
1233
+ "grad_norm": 0.18222008645534515,
1234
+ "learning_rate": 0.0009554595521988423,
1235
+ "loss": 2.934,
1236
+ "step": 15800
1237
+ },
1238
+ {
1239
+ "epoch": 0.2064935064935065,
1240
+ "grad_norm": 0.18934360146522522,
1241
+ "learning_rate": 0.0009548628114223989,
1242
+ "loss": 2.902,
1243
+ "step": 15900
1244
+ },
1245
+ {
1246
+ "epoch": 0.2077922077922078,
1247
+ "grad_norm": 0.17503845691680908,
1248
+ "learning_rate": 0.0009542622886121486,
1249
+ "loss": 2.9485,
1250
+ "step": 16000
1251
+ },
1252
+ {
1253
+ "epoch": 0.2077922077922078,
1254
+ "eval_loss": 3.305332660675049,
1255
+ "eval_runtime": 14.5112,
1256
+ "eval_samples_per_second": 39.693,
1257
+ "eval_steps_per_second": 9.923,
1258
+ "step": 16000
1259
  }
1260
  ],
1261
  "logging_steps": 100,
 
1275
  "attributes": {}
1276
  }
1277
  },
1278
+ "total_flos": 4.7630902099968e+17,
1279
  "train_batch_size": 22,
1280
  "trial_name": null,
1281
  "trial_params": null