devoppro commited on
Commit
da8f2d6
·
verified ·
1 Parent(s): a5a575e

Training in progress, step 1900, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e5057c9842c32a69d89842ae1ea0f94292f62299087952cbb292a8939dab162b
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67ce32f079c2d9383c97ce4438a5a5face034e7833873ae3afffd266f5462b3e
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0be90ec704029e3eaf3dc6c2072631e4c181bf8dfa918714f1b12040f44358bb
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:718a0f3db00824213036a2c0441849791319b7d9cf189065873bb26a7020738e
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e26e61de744c7fd53adae37853d7270aa3e689b6a223071467906f521252f455
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e570f6dc930829ccb55ab7b4594c42b0febbb8bfc105ca4997eb5887dbba1959
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.002,
6
  "eval_steps": 500,
7
- "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1128,6 +1128,216 @@
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1131
  }
1132
  ],
1133
  "logging_steps": 10,
@@ -1147,7 +1357,7 @@
1147
  "attributes": {}
1148
  }
1149
  },
1150
- "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 3.0018,
6
  "eval_steps": 500,
7
+ "global_step": 1900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
1131
+ },
1132
+ {
1133
+ "epoch": 1.0002,
1134
+ "grad_norm": 0.0,
1135
+ "learning_rate": 0.0002909278557114228,
1136
+ "loss": 95.5510498046875,
1137
+ "step": 1610
1138
+ },
1139
+ {
1140
+ "epoch": 1.0004,
1141
+ "grad_norm": 0.0,
1142
+ "learning_rate": 0.00029086773547094184,
1143
+ "loss": 98.69436645507812,
1144
+ "step": 1620
1145
+ },
1146
+ {
1147
+ "epoch": 1.0006,
1148
+ "grad_norm": 0.0,
1149
+ "learning_rate": 0.0002908076152304609,
1150
+ "loss": 99.0325439453125,
1151
+ "step": 1630
1152
+ },
1153
+ {
1154
+ "epoch": 1.0008,
1155
+ "grad_norm": 0.0,
1156
+ "learning_rate": 0.00029074749498997994,
1157
+ "loss": 97.31904296875,
1158
+ "step": 1640
1159
+ },
1160
+ {
1161
+ "epoch": 1.001,
1162
+ "grad_norm": 0.0,
1163
+ "learning_rate": 0.000290687374749499,
1164
+ "loss": 98.40114135742188,
1165
+ "step": 1650
1166
+ },
1167
+ {
1168
+ "epoch": 1.0012,
1169
+ "grad_norm": 0.0,
1170
+ "learning_rate": 0.00029062725450901803,
1171
+ "loss": 99.31021728515626,
1172
+ "step": 1660
1173
+ },
1174
+ {
1175
+ "epoch": 1.0014,
1176
+ "grad_norm": 0.0,
1177
+ "learning_rate": 0.00029056713426853703,
1178
+ "loss": 99.4878662109375,
1179
+ "step": 1670
1180
+ },
1181
+ {
1182
+ "epoch": 1.0016,
1183
+ "grad_norm": 0.0,
1184
+ "learning_rate": 0.0002905070140280561,
1185
+ "loss": 99.3884521484375,
1186
+ "step": 1680
1187
+ },
1188
+ {
1189
+ "epoch": 1.0018,
1190
+ "grad_norm": 0.0,
1191
+ "learning_rate": 0.0002904468937875751,
1192
+ "loss": 99.46731567382812,
1193
+ "step": 1690
1194
+ },
1195
+ {
1196
+ "epoch": 1.002,
1197
+ "grad_norm": 0.0,
1198
+ "learning_rate": 0.00029038677354709417,
1199
+ "loss": 99.4464599609375,
1200
+ "step": 1700
1201
+ },
1202
+ {
1203
+ "epoch": 2.0001,
1204
+ "grad_norm": 0.0,
1205
+ "learning_rate": 0.00029032665330661317,
1206
+ "loss": 104.66427001953124,
1207
+ "step": 1710
1208
+ },
1209
+ {
1210
+ "epoch": 2.0003,
1211
+ "grad_norm": 0.0,
1212
+ "learning_rate": 0.0002902665330661322,
1213
+ "loss": 98.24152221679688,
1214
+ "step": 1720
1215
+ },
1216
+ {
1217
+ "epoch": 2.0005,
1218
+ "grad_norm": 0.0,
1219
+ "learning_rate": 0.0002902064128256513,
1220
+ "loss": 99.4651123046875,
1221
+ "step": 1730
1222
+ },
1223
+ {
1224
+ "epoch": 2.0007,
1225
+ "grad_norm": 0.0,
1226
+ "learning_rate": 0.0002901462925851703,
1227
+ "loss": 99.3810546875,
1228
+ "step": 1740
1229
+ },
1230
+ {
1231
+ "epoch": 2.0009,
1232
+ "grad_norm": 0.0,
1233
+ "learning_rate": 0.00029008617234468936,
1234
+ "loss": 96.20846557617188,
1235
+ "step": 1750
1236
+ },
1237
+ {
1238
+ "epoch": 2.0011,
1239
+ "grad_norm": 0.0,
1240
+ "learning_rate": 0.0002900260521042084,
1241
+ "loss": 99.44166259765625,
1242
+ "step": 1760
1243
+ },
1244
+ {
1245
+ "epoch": 2.0013,
1246
+ "grad_norm": 0.0,
1247
+ "learning_rate": 0.00028996593186372745,
1248
+ "loss": 99.43170776367188,
1249
+ "step": 1770
1250
+ },
1251
+ {
1252
+ "epoch": 2.0015,
1253
+ "grad_norm": 0.0,
1254
+ "learning_rate": 0.00028990581162324644,
1255
+ "loss": 99.37213134765625,
1256
+ "step": 1780
1257
+ },
1258
+ {
1259
+ "epoch": 2.0017,
1260
+ "grad_norm": 0.0,
1261
+ "learning_rate": 0.0002898456913827655,
1262
+ "loss": 99.51829223632812,
1263
+ "step": 1790
1264
+ },
1265
+ {
1266
+ "epoch": 2.0019,
1267
+ "grad_norm": 0.0,
1268
+ "learning_rate": 0.00028978557114228454,
1269
+ "loss": 99.27789306640625,
1270
+ "step": 1800
1271
+ },
1272
+ {
1273
+ "epoch": 2.0021,
1274
+ "grad_norm": 0.0,
1275
+ "learning_rate": 0.0002897254509018036,
1276
+ "loss": 99.44300537109375,
1277
+ "step": 1810
1278
+ },
1279
+ {
1280
+ "epoch": 3.0002,
1281
+ "grad_norm": 0.0,
1282
+ "learning_rate": 0.00028966533066132264,
1283
+ "loss": 103.58494873046875,
1284
+ "step": 1820
1285
+ },
1286
+ {
1287
+ "epoch": 3.0004,
1288
+ "grad_norm": 0.0,
1289
+ "learning_rate": 0.0002896052104208417,
1290
+ "loss": 99.35955810546875,
1291
+ "step": 1830
1292
+ },
1293
+ {
1294
+ "epoch": 3.0006,
1295
+ "grad_norm": 0.0,
1296
+ "learning_rate": 0.00028954509018036073,
1297
+ "loss": 99.25761108398437,
1298
+ "step": 1840
1299
+ },
1300
+ {
1301
+ "epoch": 3.0008,
1302
+ "grad_norm": 0.0,
1303
+ "learning_rate": 0.0002894849699398797,
1304
+ "loss": 97.34839477539063,
1305
+ "step": 1850
1306
+ },
1307
+ {
1308
+ "epoch": 3.001,
1309
+ "grad_norm": 0.0,
1310
+ "learning_rate": 0.00028942484969939877,
1311
+ "loss": 98.359326171875,
1312
+ "step": 1860
1313
+ },
1314
+ {
1315
+ "epoch": 3.0012,
1316
+ "grad_norm": 0.0,
1317
+ "learning_rate": 0.0002893647294589178,
1318
+ "loss": 99.25703735351563,
1319
+ "step": 1870
1320
+ },
1321
+ {
1322
+ "epoch": 3.0014,
1323
+ "grad_norm": 0.0,
1324
+ "learning_rate": 0.00028930460921843687,
1325
+ "loss": 99.42723388671875,
1326
+ "step": 1880
1327
+ },
1328
+ {
1329
+ "epoch": 3.0016,
1330
+ "grad_norm": 0.0,
1331
+ "learning_rate": 0.00028924448897795586,
1332
+ "loss": 99.32404174804688,
1333
+ "step": 1890
1334
+ },
1335
+ {
1336
+ "epoch": 3.0018,
1337
+ "grad_norm": 0.0,
1338
+ "learning_rate": 0.0002891843687374749,
1339
+ "loss": 99.3990966796875,
1340
+ "step": 1900
1341
  }
1342
  ],
1343
  "logging_steps": 10,
 
1357
  "attributes": {}
1358
  }
1359
  },
1360
+ "total_flos": 2.680797189666816e+16,
1361
  "train_batch_size": 2,
1362
  "trial_name": null,
1363
  "trial_params": null