CodeIsAbstract commited on
Commit
7102755
·
verified ·
1 Parent(s): 4360927

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:89952de3671811784fe7b357f32398d8262fdfa34181cc91ec5b8a5baba9ebc3
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bce9cd2ecb7520c5fbf1827487189abe146f8a04937f02a2e35fa41e6c94ae73
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0192b698deb59413cfa00577b1ee14f0ef3ba0e14eb3aa418fb3f772003b2fd3
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a77095985c695d5aefdce15118c41ee7d9a582af6b20e0ad1dd0d0561b91b3f
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8c2ed9303a8f4e0d19061969cd3c3a9b1a1fd302644d59f1e023ca72324b1d0
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3136c830d048cb66eafa21e920652967c1c1c1766cfb212fe4e06f8320840cf4
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:793c5540f8117a4ddd149538354eaaa33f8c7f011c83883ddf29fe821afe64fa
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a6e444c46ec49de792e4afbe9af4aa4613bca60425da2b0ac2cae225e516fcc
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.3,
6
  "eval_steps": 100,
7
- "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1178,6 +1178,396 @@
1178
  "eval_samples_per_second": 67.998,
1179
  "eval_steps_per_second": 3.879,
1180
  "step": 1500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1181
  }
1182
  ],
1183
  "logging_steps": 10,
@@ -1197,7 +1587,7 @@
1197
  "attributes": {}
1198
  }
1199
  },
1200
- "total_flos": 4.89231772286976e+17,
1201
  "train_batch_size": 18,
1202
  "trial_name": null,
1203
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 100,
7
+ "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1178
  "eval_samples_per_second": 67.998,
1179
  "eval_steps_per_second": 3.879,
1180
  "step": 1500
1181
+ },
1182
+ {
1183
+ "epoch": 0.302,
1184
+ "grad_norm": 0.058349609375,
1185
+ "learning_rate": 0.0003,
1186
+ "loss": 2.601143646240234,
1187
+ "step": 1510
1188
+ },
1189
+ {
1190
+ "epoch": 0.304,
1191
+ "grad_norm": 0.177734375,
1192
+ "learning_rate": 0.0003,
1193
+ "loss": 2.6228967666625977,
1194
+ "step": 1520
1195
+ },
1196
+ {
1197
+ "epoch": 0.306,
1198
+ "grad_norm": 0.12158203125,
1199
+ "learning_rate": 0.0003,
1200
+ "loss": 2.611318588256836,
1201
+ "step": 1530
1202
+ },
1203
+ {
1204
+ "epoch": 0.308,
1205
+ "grad_norm": 0.06201171875,
1206
+ "learning_rate": 0.0003,
1207
+ "loss": 2.61364860534668,
1208
+ "step": 1540
1209
+ },
1210
+ {
1211
+ "epoch": 0.31,
1212
+ "grad_norm": 0.05419921875,
1213
+ "learning_rate": 0.0003,
1214
+ "loss": 2.6139429092407225,
1215
+ "step": 1550
1216
+ },
1217
+ {
1218
+ "epoch": 0.312,
1219
+ "grad_norm": 0.08837890625,
1220
+ "learning_rate": 0.0003,
1221
+ "loss": 2.6160503387451173,
1222
+ "step": 1560
1223
+ },
1224
+ {
1225
+ "epoch": 0.314,
1226
+ "grad_norm": 0.0673828125,
1227
+ "learning_rate": 0.0003,
1228
+ "loss": 2.615163230895996,
1229
+ "step": 1570
1230
+ },
1231
+ {
1232
+ "epoch": 0.316,
1233
+ "grad_norm": 0.055419921875,
1234
+ "learning_rate": 0.0003,
1235
+ "loss": 2.592479705810547,
1236
+ "step": 1580
1237
+ },
1238
+ {
1239
+ "epoch": 0.318,
1240
+ "grad_norm": 0.41796875,
1241
+ "learning_rate": 0.0003,
1242
+ "loss": 2.587255668640137,
1243
+ "step": 1590
1244
+ },
1245
+ {
1246
+ "epoch": 0.32,
1247
+ "grad_norm": 0.11181640625,
1248
+ "learning_rate": 0.0003,
1249
+ "loss": 2.6103662490844726,
1250
+ "step": 1600
1251
+ },
1252
+ {
1253
+ "epoch": 0.32,
1254
+ "eval_loss": 3.0303030014038086,
1255
+ "eval_runtime": 4.4554,
1256
+ "eval_samples_per_second": 66.885,
1257
+ "eval_steps_per_second": 3.816,
1258
+ "step": 1600
1259
+ },
1260
+ {
1261
+ "epoch": 0.322,
1262
+ "grad_norm": 0.056884765625,
1263
+ "learning_rate": 0.0003,
1264
+ "loss": 2.602743911743164,
1265
+ "step": 1610
1266
+ },
1267
+ {
1268
+ "epoch": 0.324,
1269
+ "grad_norm": 0.08642578125,
1270
+ "learning_rate": 0.0003,
1271
+ "loss": 2.604319953918457,
1272
+ "step": 1620
1273
+ },
1274
+ {
1275
+ "epoch": 0.326,
1276
+ "grad_norm": 0.5859375,
1277
+ "learning_rate": 0.0003,
1278
+ "loss": 2.602446174621582,
1279
+ "step": 1630
1280
+ },
1281
+ {
1282
+ "epoch": 0.328,
1283
+ "grad_norm": 0.056396484375,
1284
+ "learning_rate": 0.0003,
1285
+ "loss": 2.6054025650024415,
1286
+ "step": 1640
1287
+ },
1288
+ {
1289
+ "epoch": 0.33,
1290
+ "grad_norm": 0.0791015625,
1291
+ "learning_rate": 0.0003,
1292
+ "loss": 2.597785758972168,
1293
+ "step": 1650
1294
+ },
1295
+ {
1296
+ "epoch": 0.332,
1297
+ "grad_norm": 0.1259765625,
1298
+ "learning_rate": 0.0003,
1299
+ "loss": 2.5736499786376954,
1300
+ "step": 1660
1301
+ },
1302
+ {
1303
+ "epoch": 0.334,
1304
+ "grad_norm": 0.07470703125,
1305
+ "learning_rate": 0.0003,
1306
+ "loss": 2.5932741165161133,
1307
+ "step": 1670
1308
+ },
1309
+ {
1310
+ "epoch": 0.336,
1311
+ "grad_norm": 0.058837890625,
1312
+ "learning_rate": 0.0003,
1313
+ "loss": 2.6116256713867188,
1314
+ "step": 1680
1315
+ },
1316
+ {
1317
+ "epoch": 0.338,
1318
+ "grad_norm": 0.057861328125,
1319
+ "learning_rate": 0.0003,
1320
+ "loss": 2.625630187988281,
1321
+ "step": 1690
1322
+ },
1323
+ {
1324
+ "epoch": 0.34,
1325
+ "grad_norm": 0.06640625,
1326
+ "learning_rate": 0.0003,
1327
+ "loss": 2.596027374267578,
1328
+ "step": 1700
1329
+ },
1330
+ {
1331
+ "epoch": 0.34,
1332
+ "eval_loss": 3.028986692428589,
1333
+ "eval_runtime": 4.4483,
1334
+ "eval_samples_per_second": 66.992,
1335
+ "eval_steps_per_second": 3.822,
1336
+ "step": 1700
1337
+ },
1338
+ {
1339
+ "epoch": 0.342,
1340
+ "grad_norm": 0.25,
1341
+ "learning_rate": 0.0003,
1342
+ "loss": 2.6015995025634764,
1343
+ "step": 1710
1344
+ },
1345
+ {
1346
+ "epoch": 0.344,
1347
+ "grad_norm": 0.057373046875,
1348
+ "learning_rate": 0.0003,
1349
+ "loss": 2.598787307739258,
1350
+ "step": 1720
1351
+ },
1352
+ {
1353
+ "epoch": 0.346,
1354
+ "grad_norm": 0.08544921875,
1355
+ "learning_rate": 0.0003,
1356
+ "loss": 2.6199289321899415,
1357
+ "step": 1730
1358
+ },
1359
+ {
1360
+ "epoch": 0.348,
1361
+ "grad_norm": 0.060302734375,
1362
+ "learning_rate": 0.0003,
1363
+ "loss": 2.602091598510742,
1364
+ "step": 1740
1365
+ },
1366
+ {
1367
+ "epoch": 0.35,
1368
+ "grad_norm": 4.875,
1369
+ "learning_rate": 0.0003,
1370
+ "loss": 2.5984159469604493,
1371
+ "step": 1750
1372
+ },
1373
+ {
1374
+ "epoch": 0.352,
1375
+ "grad_norm": 0.05224609375,
1376
+ "learning_rate": 0.0003,
1377
+ "loss": 2.6147043228149416,
1378
+ "step": 1760
1379
+ },
1380
+ {
1381
+ "epoch": 0.354,
1382
+ "grad_norm": 0.1982421875,
1383
+ "learning_rate": 0.0003,
1384
+ "loss": 2.6081768035888673,
1385
+ "step": 1770
1386
+ },
1387
+ {
1388
+ "epoch": 0.356,
1389
+ "grad_norm": 0.1328125,
1390
+ "learning_rate": 0.0003,
1391
+ "loss": 2.6030120849609375,
1392
+ "step": 1780
1393
+ },
1394
+ {
1395
+ "epoch": 0.358,
1396
+ "grad_norm": 0.08642578125,
1397
+ "learning_rate": 0.0003,
1398
+ "loss": 2.6104217529296876,
1399
+ "step": 1790
1400
+ },
1401
+ {
1402
+ "epoch": 0.36,
1403
+ "grad_norm": 0.051025390625,
1404
+ "learning_rate": 0.0003,
1405
+ "loss": 2.5706939697265625,
1406
+ "step": 1800
1407
+ },
1408
+ {
1409
+ "epoch": 0.36,
1410
+ "eval_loss": 3.0339813232421875,
1411
+ "eval_runtime": 4.452,
1412
+ "eval_samples_per_second": 66.937,
1413
+ "eval_steps_per_second": 3.819,
1414
+ "step": 1800
1415
+ },
1416
+ {
1417
+ "epoch": 0.362,
1418
+ "grad_norm": 0.1376953125,
1419
+ "learning_rate": 0.0003,
1420
+ "loss": 2.576882743835449,
1421
+ "step": 1810
1422
+ },
1423
+ {
1424
+ "epoch": 0.364,
1425
+ "grad_norm": 0.1162109375,
1426
+ "learning_rate": 0.0003,
1427
+ "loss": 2.601199913024902,
1428
+ "step": 1820
1429
+ },
1430
+ {
1431
+ "epoch": 0.366,
1432
+ "grad_norm": 0.062255859375,
1433
+ "learning_rate": 0.0003,
1434
+ "loss": 2.570973777770996,
1435
+ "step": 1830
1436
+ },
1437
+ {
1438
+ "epoch": 0.368,
1439
+ "grad_norm": 0.05859375,
1440
+ "learning_rate": 0.0003,
1441
+ "loss": 2.618685722351074,
1442
+ "step": 1840
1443
+ },
1444
+ {
1445
+ "epoch": 0.37,
1446
+ "grad_norm": 0.05859375,
1447
+ "learning_rate": 0.0003,
1448
+ "loss": 2.613678550720215,
1449
+ "step": 1850
1450
+ },
1451
+ {
1452
+ "epoch": 0.372,
1453
+ "grad_norm": 0.06640625,
1454
+ "learning_rate": 0.0003,
1455
+ "loss": 2.595960235595703,
1456
+ "step": 1860
1457
+ },
1458
+ {
1459
+ "epoch": 0.374,
1460
+ "grad_norm": 0.0595703125,
1461
+ "learning_rate": 0.0003,
1462
+ "loss": 2.6117223739624023,
1463
+ "step": 1870
1464
+ },
1465
+ {
1466
+ "epoch": 0.376,
1467
+ "grad_norm": 0.09423828125,
1468
+ "learning_rate": 0.0003,
1469
+ "loss": 2.5977685928344725,
1470
+ "step": 1880
1471
+ },
1472
+ {
1473
+ "epoch": 0.378,
1474
+ "grad_norm": 0.060791015625,
1475
+ "learning_rate": 0.0003,
1476
+ "loss": 2.619519805908203,
1477
+ "step": 1890
1478
+ },
1479
+ {
1480
+ "epoch": 0.38,
1481
+ "grad_norm": 0.12060546875,
1482
+ "learning_rate": 0.0003,
1483
+ "loss": 2.606584930419922,
1484
+ "step": 1900
1485
+ },
1486
+ {
1487
+ "epoch": 0.38,
1488
+ "eval_loss": 3.0274596214294434,
1489
+ "eval_runtime": 4.4258,
1490
+ "eval_samples_per_second": 67.332,
1491
+ "eval_steps_per_second": 3.841,
1492
+ "step": 1900
1493
+ },
1494
+ {
1495
+ "epoch": 0.382,
1496
+ "grad_norm": 0.052734375,
1497
+ "learning_rate": 0.0003,
1498
+ "loss": 2.598057174682617,
1499
+ "step": 1910
1500
+ },
1501
+ {
1502
+ "epoch": 0.384,
1503
+ "grad_norm": 0.09228515625,
1504
+ "learning_rate": 0.0003,
1505
+ "loss": 2.5836353302001953,
1506
+ "step": 1920
1507
+ },
1508
+ {
1509
+ "epoch": 0.386,
1510
+ "grad_norm": 0.68359375,
1511
+ "learning_rate": 0.0003,
1512
+ "loss": 2.5969900131225585,
1513
+ "step": 1930
1514
+ },
1515
+ {
1516
+ "epoch": 0.388,
1517
+ "grad_norm": 0.06982421875,
1518
+ "learning_rate": 0.0003,
1519
+ "loss": 2.610474395751953,
1520
+ "step": 1940
1521
+ },
1522
+ {
1523
+ "epoch": 0.39,
1524
+ "grad_norm": 0.16015625,
1525
+ "learning_rate": 0.0003,
1526
+ "loss": 2.589932441711426,
1527
+ "step": 1950
1528
+ },
1529
+ {
1530
+ "epoch": 0.392,
1531
+ "grad_norm": 0.05908203125,
1532
+ "learning_rate": 0.0003,
1533
+ "loss": 2.616892433166504,
1534
+ "step": 1960
1535
+ },
1536
+ {
1537
+ "epoch": 0.394,
1538
+ "grad_norm": 0.054931640625,
1539
+ "learning_rate": 0.0003,
1540
+ "loss": 2.5918600082397463,
1541
+ "step": 1970
1542
+ },
1543
+ {
1544
+ "epoch": 0.396,
1545
+ "grad_norm": 0.1064453125,
1546
+ "learning_rate": 0.0003,
1547
+ "loss": 2.6085411071777345,
1548
+ "step": 1980
1549
+ },
1550
+ {
1551
+ "epoch": 0.398,
1552
+ "grad_norm": 0.0576171875,
1553
+ "learning_rate": 0.0003,
1554
+ "loss": 2.598824691772461,
1555
+ "step": 1990
1556
+ },
1557
+ {
1558
+ "epoch": 0.4,
1559
+ "grad_norm": 0.07861328125,
1560
+ "learning_rate": 0.0003,
1561
+ "loss": 2.5938999176025392,
1562
+ "step": 2000
1563
+ },
1564
+ {
1565
+ "epoch": 0.4,
1566
+ "eval_loss": 3.030641794204712,
1567
+ "eval_runtime": 4.4468,
1568
+ "eval_samples_per_second": 67.014,
1569
+ "eval_steps_per_second": 3.823,
1570
+ "step": 2000
1571
  }
1572
  ],
1573
  "logging_steps": 10,
 
1587
  "attributes": {}
1588
  }
1589
  },
1590
+ "total_flos": 6.52309029715968e+17,
1591
  "train_batch_size": 18,
1592
  "trial_name": null,
1593
  "trial_params": null