devoppro commited on
Commit
82cc161
·
verified ·
1 Parent(s): f79ea23

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e5057c9842c32a69d89842ae1ea0f94292f62299087952cbb292a8939dab162b
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd1f6e22634fcac871f4ed24bc640ad73cdede31b6d8fa36f1bdebd04d0ef694
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c03cf43380d82cadaa6b3a05aeb813bcd9a1b679482ca2dbec8dd79ee35ffa3
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01f9a0f7843a37be87edd23f4e88aa93b38b95cc2c07503eeb1cf2e4632453a2
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:946093fecc07b9eebf6c92eb089b5a00988fc00e39f09c111679745018d0d6ac
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8af7bf0eac60970e638b41829ebcbe1a8ce90393891b1a1fef11c365fd9582d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.002,
6
  "eval_steps": 500,
7
- "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1128,6 +1128,286 @@
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1131
  }
1132
  ],
1133
  "logging_steps": 10,
@@ -1147,7 +1427,7 @@
1147
  "attributes": {}
1148
  }
1149
  },
1150
- "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 4.0017,
6
  "eval_steps": 500,
7
+ "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
1131
+ },
1132
+ {
1133
+ "epoch": 1.0002,
1134
+ "grad_norm": 0.0,
1135
+ "learning_rate": 0.0002909278557114228,
1136
+ "loss": 95.5510498046875,
1137
+ "step": 1610
1138
+ },
1139
+ {
1140
+ "epoch": 1.0004,
1141
+ "grad_norm": 0.0,
1142
+ "learning_rate": 0.00029086773547094184,
1143
+ "loss": 98.69436645507812,
1144
+ "step": 1620
1145
+ },
1146
+ {
1147
+ "epoch": 1.0006,
1148
+ "grad_norm": 0.0,
1149
+ "learning_rate": 0.0002908076152304609,
1150
+ "loss": 99.0325439453125,
1151
+ "step": 1630
1152
+ },
1153
+ {
1154
+ "epoch": 1.0008,
1155
+ "grad_norm": 0.0,
1156
+ "learning_rate": 0.00029074749498997994,
1157
+ "loss": 97.31904296875,
1158
+ "step": 1640
1159
+ },
1160
+ {
1161
+ "epoch": 1.001,
1162
+ "grad_norm": 0.0,
1163
+ "learning_rate": 0.000290687374749499,
1164
+ "loss": 98.40114135742188,
1165
+ "step": 1650
1166
+ },
1167
+ {
1168
+ "epoch": 1.0012,
1169
+ "grad_norm": 0.0,
1170
+ "learning_rate": 0.00029062725450901803,
1171
+ "loss": 99.31021728515626,
1172
+ "step": 1660
1173
+ },
1174
+ {
1175
+ "epoch": 1.0014,
1176
+ "grad_norm": 0.0,
1177
+ "learning_rate": 0.00029056713426853703,
1178
+ "loss": 99.4878662109375,
1179
+ "step": 1670
1180
+ },
1181
+ {
1182
+ "epoch": 1.0016,
1183
+ "grad_norm": 0.0,
1184
+ "learning_rate": 0.0002905070140280561,
1185
+ "loss": 99.3884521484375,
1186
+ "step": 1680
1187
+ },
1188
+ {
1189
+ "epoch": 1.0018,
1190
+ "grad_norm": 0.0,
1191
+ "learning_rate": 0.0002904468937875751,
1192
+ "loss": 99.46731567382812,
1193
+ "step": 1690
1194
+ },
1195
+ {
1196
+ "epoch": 1.002,
1197
+ "grad_norm": 0.0,
1198
+ "learning_rate": 0.00029038677354709417,
1199
+ "loss": 99.4464599609375,
1200
+ "step": 1700
1201
+ },
1202
+ {
1203
+ "epoch": 2.0001,
1204
+ "grad_norm": 0.0,
1205
+ "learning_rate": 0.00029032665330661317,
1206
+ "loss": 104.66427001953124,
1207
+ "step": 1710
1208
+ },
1209
+ {
1210
+ "epoch": 2.0003,
1211
+ "grad_norm": 0.0,
1212
+ "learning_rate": 0.0002902665330661322,
1213
+ "loss": 98.24152221679688,
1214
+ "step": 1720
1215
+ },
1216
+ {
1217
+ "epoch": 2.0005,
1218
+ "grad_norm": 0.0,
1219
+ "learning_rate": 0.0002902064128256513,
1220
+ "loss": 99.4651123046875,
1221
+ "step": 1730
1222
+ },
1223
+ {
1224
+ "epoch": 2.0007,
1225
+ "grad_norm": 0.0,
1226
+ "learning_rate": 0.0002901462925851703,
1227
+ "loss": 99.3810546875,
1228
+ "step": 1740
1229
+ },
1230
+ {
1231
+ "epoch": 2.0009,
1232
+ "grad_norm": 0.0,
1233
+ "learning_rate": 0.00029008617234468936,
1234
+ "loss": 96.20846557617188,
1235
+ "step": 1750
1236
+ },
1237
+ {
1238
+ "epoch": 2.0011,
1239
+ "grad_norm": 0.0,
1240
+ "learning_rate": 0.0002900260521042084,
1241
+ "loss": 99.44166259765625,
1242
+ "step": 1760
1243
+ },
1244
+ {
1245
+ "epoch": 2.0013,
1246
+ "grad_norm": 0.0,
1247
+ "learning_rate": 0.00028996593186372745,
1248
+ "loss": 99.43170776367188,
1249
+ "step": 1770
1250
+ },
1251
+ {
1252
+ "epoch": 2.0015,
1253
+ "grad_norm": 0.0,
1254
+ "learning_rate": 0.00028990581162324644,
1255
+ "loss": 99.37213134765625,
1256
+ "step": 1780
1257
+ },
1258
+ {
1259
+ "epoch": 2.0017,
1260
+ "grad_norm": 0.0,
1261
+ "learning_rate": 0.0002898456913827655,
1262
+ "loss": 99.51829223632812,
1263
+ "step": 1790
1264
+ },
1265
+ {
1266
+ "epoch": 2.0019,
1267
+ "grad_norm": 0.0,
1268
+ "learning_rate": 0.00028978557114228454,
1269
+ "loss": 99.27789306640625,
1270
+ "step": 1800
1271
+ },
1272
+ {
1273
+ "epoch": 2.0021,
1274
+ "grad_norm": 0.0,
1275
+ "learning_rate": 0.0002897254509018036,
1276
+ "loss": 99.44300537109375,
1277
+ "step": 1810
1278
+ },
1279
+ {
1280
+ "epoch": 3.0002,
1281
+ "grad_norm": 0.0,
1282
+ "learning_rate": 0.00028966533066132264,
1283
+ "loss": 103.58494873046875,
1284
+ "step": 1820
1285
+ },
1286
+ {
1287
+ "epoch": 3.0004,
1288
+ "grad_norm": 0.0,
1289
+ "learning_rate": 0.0002896052104208417,
1290
+ "loss": 99.35955810546875,
1291
+ "step": 1830
1292
+ },
1293
+ {
1294
+ "epoch": 3.0006,
1295
+ "grad_norm": 0.0,
1296
+ "learning_rate": 0.00028954509018036073,
1297
+ "loss": 99.25761108398437,
1298
+ "step": 1840
1299
+ },
1300
+ {
1301
+ "epoch": 3.0008,
1302
+ "grad_norm": 0.0,
1303
+ "learning_rate": 0.0002894849699398797,
1304
+ "loss": 97.34839477539063,
1305
+ "step": 1850
1306
+ },
1307
+ {
1308
+ "epoch": 3.001,
1309
+ "grad_norm": 0.0,
1310
+ "learning_rate": 0.00028942484969939877,
1311
+ "loss": 98.359326171875,
1312
+ "step": 1860
1313
+ },
1314
+ {
1315
+ "epoch": 3.0012,
1316
+ "grad_norm": 0.0,
1317
+ "learning_rate": 0.0002893647294589178,
1318
+ "loss": 99.25703735351563,
1319
+ "step": 1870
1320
+ },
1321
+ {
1322
+ "epoch": 3.0014,
1323
+ "grad_norm": 0.0,
1324
+ "learning_rate": 0.00028930460921843687,
1325
+ "loss": 99.42723388671875,
1326
+ "step": 1880
1327
+ },
1328
+ {
1329
+ "epoch": 3.0016,
1330
+ "grad_norm": 0.0,
1331
+ "learning_rate": 0.00028924448897795586,
1332
+ "loss": 99.32404174804688,
1333
+ "step": 1890
1334
+ },
1335
+ {
1336
+ "epoch": 3.0018,
1337
+ "grad_norm": 0.0,
1338
+ "learning_rate": 0.0002891843687374749,
1339
+ "loss": 99.3990966796875,
1340
+ "step": 1900
1341
+ },
1342
+ {
1343
+ "epoch": 3.002,
1344
+ "grad_norm": 0.0,
1345
+ "learning_rate": 0.00028912424849699396,
1346
+ "loss": 99.39078369140626,
1347
+ "step": 1910
1348
+ },
1349
+ {
1350
+ "epoch": 4.0001,
1351
+ "grad_norm": 0.0,
1352
+ "learning_rate": 0.000289064128256513,
1353
+ "loss": 104.5905517578125,
1354
+ "step": 1920
1355
+ },
1356
+ {
1357
+ "epoch": 4.0003,
1358
+ "grad_norm": 0.0,
1359
+ "learning_rate": 0.00028900400801603205,
1360
+ "loss": 98.17286987304688,
1361
+ "step": 1930
1362
+ },
1363
+ {
1364
+ "epoch": 4.0005,
1365
+ "grad_norm": 0.0,
1366
+ "learning_rate": 0.0002889438877755511,
1367
+ "loss": 99.40567016601562,
1368
+ "step": 1940
1369
+ },
1370
+ {
1371
+ "epoch": 4.0007,
1372
+ "grad_norm": 0.0,
1373
+ "learning_rate": 0.00028888376753507015,
1374
+ "loss": 99.32139892578125,
1375
+ "step": 1950
1376
+ },
1377
+ {
1378
+ "epoch": 4.0009,
1379
+ "grad_norm": 0.0,
1380
+ "learning_rate": 0.00028882364729458914,
1381
+ "loss": 96.15136108398437,
1382
+ "step": 1960
1383
+ },
1384
+ {
1385
+ "epoch": 4.0011,
1386
+ "grad_norm": 0.0,
1387
+ "learning_rate": 0.0002887635270541082,
1388
+ "loss": 99.3803466796875,
1389
+ "step": 1970
1390
+ },
1391
+ {
1392
+ "epoch": 4.0013,
1393
+ "grad_norm": 0.0,
1394
+ "learning_rate": 0.00028870340681362724,
1395
+ "loss": 99.37009887695312,
1396
+ "step": 1980
1397
+ },
1398
+ {
1399
+ "epoch": 4.0015,
1400
+ "grad_norm": 0.0,
1401
+ "learning_rate": 0.0002886432865731463,
1402
+ "loss": 99.310498046875,
1403
+ "step": 1990
1404
+ },
1405
+ {
1406
+ "epoch": 4.0017,
1407
+ "grad_norm": 0.0,
1408
+ "learning_rate": 0.0002885831663326653,
1409
+ "loss": 99.45986938476562,
1410
+ "step": 2000
1411
  }
1412
  ],
1413
  "logging_steps": 10,
 
1427
  "attributes": {}
1428
  }
1429
  },
1430
+ "total_flos": 2.687705099497728e+16,
1431
  "train_batch_size": 2,
1432
  "trial_name": null,
1433
  "trial_params": null