devoppro commited on
Commit
47c04ae
·
verified ·
1 Parent(s): 82cc161

Training in progress, step 2100

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cd1f6e22634fcac871f4ed24bc640ad73cdede31b6d8fa36f1bdebd04d0ef694
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e5057c9842c32a69d89842ae1ea0f94292f62299087952cbb292a8939dab162b
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c03cf43380d82cadaa6b3a05aeb813bcd9a1b679482ca2dbec8dd79ee35ffa3
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:01f9a0f7843a37be87edd23f4e88aa93b38b95cc2c07503eeb1cf2e4632453a2
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:946093fecc07b9eebf6c92eb089b5a00988fc00e39f09c111679745018d0d6ac
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c8af7bf0eac60970e638b41829ebcbe1a8ce90393891b1a1fef11c365fd9582d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 4.0017,
6
  "eval_steps": 500,
7
- "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1128,286 +1128,6 @@
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
1131
- },
1132
- {
1133
- "epoch": 1.0002,
1134
- "grad_norm": 0.0,
1135
- "learning_rate": 0.0002909278557114228,
1136
- "loss": 95.5510498046875,
1137
- "step": 1610
1138
- },
1139
- {
1140
- "epoch": 1.0004,
1141
- "grad_norm": 0.0,
1142
- "learning_rate": 0.00029086773547094184,
1143
- "loss": 98.69436645507812,
1144
- "step": 1620
1145
- },
1146
- {
1147
- "epoch": 1.0006,
1148
- "grad_norm": 0.0,
1149
- "learning_rate": 0.0002908076152304609,
1150
- "loss": 99.0325439453125,
1151
- "step": 1630
1152
- },
1153
- {
1154
- "epoch": 1.0008,
1155
- "grad_norm": 0.0,
1156
- "learning_rate": 0.00029074749498997994,
1157
- "loss": 97.31904296875,
1158
- "step": 1640
1159
- },
1160
- {
1161
- "epoch": 1.001,
1162
- "grad_norm": 0.0,
1163
- "learning_rate": 0.000290687374749499,
1164
- "loss": 98.40114135742188,
1165
- "step": 1650
1166
- },
1167
- {
1168
- "epoch": 1.0012,
1169
- "grad_norm": 0.0,
1170
- "learning_rate": 0.00029062725450901803,
1171
- "loss": 99.31021728515626,
1172
- "step": 1660
1173
- },
1174
- {
1175
- "epoch": 1.0014,
1176
- "grad_norm": 0.0,
1177
- "learning_rate": 0.00029056713426853703,
1178
- "loss": 99.4878662109375,
1179
- "step": 1670
1180
- },
1181
- {
1182
- "epoch": 1.0016,
1183
- "grad_norm": 0.0,
1184
- "learning_rate": 0.0002905070140280561,
1185
- "loss": 99.3884521484375,
1186
- "step": 1680
1187
- },
1188
- {
1189
- "epoch": 1.0018,
1190
- "grad_norm": 0.0,
1191
- "learning_rate": 0.0002904468937875751,
1192
- "loss": 99.46731567382812,
1193
- "step": 1690
1194
- },
1195
- {
1196
- "epoch": 1.002,
1197
- "grad_norm": 0.0,
1198
- "learning_rate": 0.00029038677354709417,
1199
- "loss": 99.4464599609375,
1200
- "step": 1700
1201
- },
1202
- {
1203
- "epoch": 2.0001,
1204
- "grad_norm": 0.0,
1205
- "learning_rate": 0.00029032665330661317,
1206
- "loss": 104.66427001953124,
1207
- "step": 1710
1208
- },
1209
- {
1210
- "epoch": 2.0003,
1211
- "grad_norm": 0.0,
1212
- "learning_rate": 0.0002902665330661322,
1213
- "loss": 98.24152221679688,
1214
- "step": 1720
1215
- },
1216
- {
1217
- "epoch": 2.0005,
1218
- "grad_norm": 0.0,
1219
- "learning_rate": 0.0002902064128256513,
1220
- "loss": 99.4651123046875,
1221
- "step": 1730
1222
- },
1223
- {
1224
- "epoch": 2.0007,
1225
- "grad_norm": 0.0,
1226
- "learning_rate": 0.0002901462925851703,
1227
- "loss": 99.3810546875,
1228
- "step": 1740
1229
- },
1230
- {
1231
- "epoch": 2.0009,
1232
- "grad_norm": 0.0,
1233
- "learning_rate": 0.00029008617234468936,
1234
- "loss": 96.20846557617188,
1235
- "step": 1750
1236
- },
1237
- {
1238
- "epoch": 2.0011,
1239
- "grad_norm": 0.0,
1240
- "learning_rate": 0.0002900260521042084,
1241
- "loss": 99.44166259765625,
1242
- "step": 1760
1243
- },
1244
- {
1245
- "epoch": 2.0013,
1246
- "grad_norm": 0.0,
1247
- "learning_rate": 0.00028996593186372745,
1248
- "loss": 99.43170776367188,
1249
- "step": 1770
1250
- },
1251
- {
1252
- "epoch": 2.0015,
1253
- "grad_norm": 0.0,
1254
- "learning_rate": 0.00028990581162324644,
1255
- "loss": 99.37213134765625,
1256
- "step": 1780
1257
- },
1258
- {
1259
- "epoch": 2.0017,
1260
- "grad_norm": 0.0,
1261
- "learning_rate": 0.0002898456913827655,
1262
- "loss": 99.51829223632812,
1263
- "step": 1790
1264
- },
1265
- {
1266
- "epoch": 2.0019,
1267
- "grad_norm": 0.0,
1268
- "learning_rate": 0.00028978557114228454,
1269
- "loss": 99.27789306640625,
1270
- "step": 1800
1271
- },
1272
- {
1273
- "epoch": 2.0021,
1274
- "grad_norm": 0.0,
1275
- "learning_rate": 0.0002897254509018036,
1276
- "loss": 99.44300537109375,
1277
- "step": 1810
1278
- },
1279
- {
1280
- "epoch": 3.0002,
1281
- "grad_norm": 0.0,
1282
- "learning_rate": 0.00028966533066132264,
1283
- "loss": 103.58494873046875,
1284
- "step": 1820
1285
- },
1286
- {
1287
- "epoch": 3.0004,
1288
- "grad_norm": 0.0,
1289
- "learning_rate": 0.0002896052104208417,
1290
- "loss": 99.35955810546875,
1291
- "step": 1830
1292
- },
1293
- {
1294
- "epoch": 3.0006,
1295
- "grad_norm": 0.0,
1296
- "learning_rate": 0.00028954509018036073,
1297
- "loss": 99.25761108398437,
1298
- "step": 1840
1299
- },
1300
- {
1301
- "epoch": 3.0008,
1302
- "grad_norm": 0.0,
1303
- "learning_rate": 0.0002894849699398797,
1304
- "loss": 97.34839477539063,
1305
- "step": 1850
1306
- },
1307
- {
1308
- "epoch": 3.001,
1309
- "grad_norm": 0.0,
1310
- "learning_rate": 0.00028942484969939877,
1311
- "loss": 98.359326171875,
1312
- "step": 1860
1313
- },
1314
- {
1315
- "epoch": 3.0012,
1316
- "grad_norm": 0.0,
1317
- "learning_rate": 0.0002893647294589178,
1318
- "loss": 99.25703735351563,
1319
- "step": 1870
1320
- },
1321
- {
1322
- "epoch": 3.0014,
1323
- "grad_norm": 0.0,
1324
- "learning_rate": 0.00028930460921843687,
1325
- "loss": 99.42723388671875,
1326
- "step": 1880
1327
- },
1328
- {
1329
- "epoch": 3.0016,
1330
- "grad_norm": 0.0,
1331
- "learning_rate": 0.00028924448897795586,
1332
- "loss": 99.32404174804688,
1333
- "step": 1890
1334
- },
1335
- {
1336
- "epoch": 3.0018,
1337
- "grad_norm": 0.0,
1338
- "learning_rate": 0.0002891843687374749,
1339
- "loss": 99.3990966796875,
1340
- "step": 1900
1341
- },
1342
- {
1343
- "epoch": 3.002,
1344
- "grad_norm": 0.0,
1345
- "learning_rate": 0.00028912424849699396,
1346
- "loss": 99.39078369140626,
1347
- "step": 1910
1348
- },
1349
- {
1350
- "epoch": 4.0001,
1351
- "grad_norm": 0.0,
1352
- "learning_rate": 0.000289064128256513,
1353
- "loss": 104.5905517578125,
1354
- "step": 1920
1355
- },
1356
- {
1357
- "epoch": 4.0003,
1358
- "grad_norm": 0.0,
1359
- "learning_rate": 0.00028900400801603205,
1360
- "loss": 98.17286987304688,
1361
- "step": 1930
1362
- },
1363
- {
1364
- "epoch": 4.0005,
1365
- "grad_norm": 0.0,
1366
- "learning_rate": 0.0002889438877755511,
1367
- "loss": 99.40567016601562,
1368
- "step": 1940
1369
- },
1370
- {
1371
- "epoch": 4.0007,
1372
- "grad_norm": 0.0,
1373
- "learning_rate": 0.00028888376753507015,
1374
- "loss": 99.32139892578125,
1375
- "step": 1950
1376
- },
1377
- {
1378
- "epoch": 4.0009,
1379
- "grad_norm": 0.0,
1380
- "learning_rate": 0.00028882364729458914,
1381
- "loss": 96.15136108398437,
1382
- "step": 1960
1383
- },
1384
- {
1385
- "epoch": 4.0011,
1386
- "grad_norm": 0.0,
1387
- "learning_rate": 0.0002887635270541082,
1388
- "loss": 99.3803466796875,
1389
- "step": 1970
1390
- },
1391
- {
1392
- "epoch": 4.0013,
1393
- "grad_norm": 0.0,
1394
- "learning_rate": 0.00028870340681362724,
1395
- "loss": 99.37009887695312,
1396
- "step": 1980
1397
- },
1398
- {
1399
- "epoch": 4.0015,
1400
- "grad_norm": 0.0,
1401
- "learning_rate": 0.0002886432865731463,
1402
- "loss": 99.310498046875,
1403
- "step": 1990
1404
- },
1405
- {
1406
- "epoch": 4.0017,
1407
- "grad_norm": 0.0,
1408
- "learning_rate": 0.0002885831663326653,
1409
- "loss": 99.45986938476562,
1410
- "step": 2000
1411
  }
1412
  ],
1413
  "logging_steps": 10,
@@ -1427,7 +1147,7 @@
1427
  "attributes": {}
1428
  }
1429
  },
1430
- "total_flos": 2.687705099497728e+16,
1431
  "train_batch_size": 2,
1432
  "trial_name": null,
1433
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.002,
6
  "eval_steps": 500,
7
+ "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1131
  }
1132
  ],
1133
  "logging_steps": 10,
 
1147
  "attributes": {}
1148
  }
1149
  },
1150
+ "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cd1f6e22634fcac871f4ed24bc640ad73cdede31b6d8fa36f1bdebd04d0ef694
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba9b7116551b80d42e6a8f1e22842f05a9e02c4e9887b27fa6e81666ef57335c
3
  size 1235573136