CodeIsAbstract commited on
Commit
f79c89e
·
verified ·
1 Parent(s): a4bb93d

Training in progress, step 400, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b0f05902edbab8c2047e2cd165f535365973f4882aa48d72f02b579256828033
3
  size 1738460416
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bbe6fe1babb0e8d8c27837f99199a1442757cd3c7e7a0ca8acb25c6876e86e6e
3
  size 1738460416
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b51c46bae4daf9e6a9e338f80a67d2671a57ba0289d103540996cd83f303338c
3
  size 3477327340
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a68596138663ffb41516fc6f90b803f499772e83b55e196c5e91745524db2517
3
  size 3477327340
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c213a373f3e5d95993ad095a3790a902d821a1b4b93a10cc7d382c8726fcb9d
3
  size 15429
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:81d5f83aeb4b3f559bd28377336d47659b320e7f6ef2e5a723d284716278a151
3
  size 15429
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8fb125336725f7741cb4daa1e3d06e225bbacfde8d41c4dcabb6762c222e62c6
3
  size 15429
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2626437dcb133ffcf003ac89603f8cce07459b93a98d760cd9419e0d6a994067
3
  size 15429
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:153c865f77c7129ba565bded50f334683d51c80f20e3cfec39e62f8737b86f0d
3
  size 15429
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae777e24d50cb7159634e1245f0697ba0fc64d5b26d535f2c80e411371a90b1c
3
  size 15429
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d017ce00fcebac7edc058ddd138f194eb0340f2d8ad0879bdab08f922ed0846e
3
  size 15429
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:afc5a67564eebcfc961e8f1406a7418cc73497c2935a39af0232ef59f8153a6a
3
  size 15429
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:043e3a8e2a2579bce7c021fb61a180d705f6161a67c5cea201876b0db210edab
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51eb67b845fb6f64dece598e11e09c3b72500e12b8629f58f93cab1700a7a3de
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.13333333333333333,
6
  "eval_steps": 100,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -304,6 +304,302 @@
304
  "eval_samples_per_second": 39.103,
305
  "eval_steps_per_second": 1.018,
306
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
307
  }
308
  ],
309
  "logging_steps": 5,
@@ -323,7 +619,7 @@
323
  "attributes": {}
324
  }
325
  },
326
- "total_flos": 1.3081469656236032e+17,
327
  "train_batch_size": 10,
328
  "trial_name": null,
329
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.26666666666666666,
6
  "eval_steps": 100,
7
+ "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
304
  "eval_samples_per_second": 39.103,
305
  "eval_steps_per_second": 1.018,
306
  "step": 200
307
+ },
308
+ {
309
+ "epoch": 0.13666666666666666,
310
+ "grad_norm": 4.0625,
311
+ "learning_rate": 2e-05,
312
+ "loss": 31.3537,
313
+ "step": 205
314
+ },
315
+ {
316
+ "epoch": 0.14,
317
+ "grad_norm": 4.34375,
318
+ "learning_rate": 2e-05,
319
+ "loss": 31.1251,
320
+ "step": 210
321
+ },
322
+ {
323
+ "epoch": 0.14333333333333334,
324
+ "grad_norm": 3.90625,
325
+ "learning_rate": 2e-05,
326
+ "loss": 31.0599,
327
+ "step": 215
328
+ },
329
+ {
330
+ "epoch": 0.14666666666666667,
331
+ "grad_norm": 3.96875,
332
+ "learning_rate": 2e-05,
333
+ "loss": 30.9825,
334
+ "step": 220
335
+ },
336
+ {
337
+ "epoch": 0.15,
338
+ "grad_norm": 3.78125,
339
+ "learning_rate": 2e-05,
340
+ "loss": 30.7881,
341
+ "step": 225
342
+ },
343
+ {
344
+ "epoch": 0.15333333333333332,
345
+ "grad_norm": 3.953125,
346
+ "learning_rate": 2e-05,
347
+ "loss": 30.6811,
348
+ "step": 230
349
+ },
350
+ {
351
+ "epoch": 0.15666666666666668,
352
+ "grad_norm": 3.609375,
353
+ "learning_rate": 2e-05,
354
+ "loss": 30.6916,
355
+ "step": 235
356
+ },
357
+ {
358
+ "epoch": 0.16,
359
+ "grad_norm": 4.1875,
360
+ "learning_rate": 2e-05,
361
+ "loss": 30.3585,
362
+ "step": 240
363
+ },
364
+ {
365
+ "epoch": 0.16333333333333333,
366
+ "grad_norm": 5.90625,
367
+ "learning_rate": 2e-05,
368
+ "loss": 30.4089,
369
+ "step": 245
370
+ },
371
+ {
372
+ "epoch": 0.16666666666666666,
373
+ "grad_norm": 15.3125,
374
+ "learning_rate": 2e-05,
375
+ "loss": 30.3882,
376
+ "step": 250
377
+ },
378
+ {
379
+ "epoch": 0.17,
380
+ "grad_norm": 8.25,
381
+ "learning_rate": 2e-05,
382
+ "loss": 30.2357,
383
+ "step": 255
384
+ },
385
+ {
386
+ "epoch": 0.17333333333333334,
387
+ "grad_norm": 8.875,
388
+ "learning_rate": 2e-05,
389
+ "loss": 30.2875,
390
+ "step": 260
391
+ },
392
+ {
393
+ "epoch": 0.17666666666666667,
394
+ "grad_norm": 21.375,
395
+ "learning_rate": 2e-05,
396
+ "loss": 30.1799,
397
+ "step": 265
398
+ },
399
+ {
400
+ "epoch": 0.18,
401
+ "grad_norm": 33.25,
402
+ "learning_rate": 2e-05,
403
+ "loss": 30.3195,
404
+ "step": 270
405
+ },
406
+ {
407
+ "epoch": 0.18333333333333332,
408
+ "grad_norm": 14.0625,
409
+ "learning_rate": 2e-05,
410
+ "loss": 30.3163,
411
+ "step": 275
412
+ },
413
+ {
414
+ "epoch": 0.18666666666666668,
415
+ "grad_norm": 5.25,
416
+ "learning_rate": 2e-05,
417
+ "loss": 30.2876,
418
+ "step": 280
419
+ },
420
+ {
421
+ "epoch": 0.19,
422
+ "grad_norm": 3.78125,
423
+ "learning_rate": 2e-05,
424
+ "loss": 30.3209,
425
+ "step": 285
426
+ },
427
+ {
428
+ "epoch": 0.19333333333333333,
429
+ "grad_norm": 3.078125,
430
+ "learning_rate": 2e-05,
431
+ "loss": 30.1198,
432
+ "step": 290
433
+ },
434
+ {
435
+ "epoch": 0.19666666666666666,
436
+ "grad_norm": 4.5625,
437
+ "learning_rate": 2e-05,
438
+ "loss": 30.0731,
439
+ "step": 295
440
+ },
441
+ {
442
+ "epoch": 0.2,
443
+ "grad_norm": 25.5,
444
+ "learning_rate": 2e-05,
445
+ "loss": 29.9634,
446
+ "step": 300
447
+ },
448
+ {
449
+ "epoch": 0.2,
450
+ "eval_loss": 7.543545246124268,
451
+ "eval_runtime": 6.7597,
452
+ "eval_samples_per_second": 39.794,
453
+ "eval_steps_per_second": 1.036,
454
+ "step": 300
455
+ },
456
+ {
457
+ "epoch": 0.20333333333333334,
458
+ "grad_norm": 17.375,
459
+ "learning_rate": 2e-05,
460
+ "loss": 29.8938,
461
+ "step": 305
462
+ },
463
+ {
464
+ "epoch": 0.20666666666666667,
465
+ "grad_norm": 24.375,
466
+ "learning_rate": 2e-05,
467
+ "loss": 29.8926,
468
+ "step": 310
469
+ },
470
+ {
471
+ "epoch": 0.21,
472
+ "grad_norm": 36.5,
473
+ "learning_rate": 2e-05,
474
+ "loss": 29.9906,
475
+ "step": 315
476
+ },
477
+ {
478
+ "epoch": 0.21333333333333335,
479
+ "grad_norm": 23.0,
480
+ "learning_rate": 2e-05,
481
+ "loss": 29.9687,
482
+ "step": 320
483
+ },
484
+ {
485
+ "epoch": 0.21666666666666667,
486
+ "grad_norm": 30.625,
487
+ "learning_rate": 2e-05,
488
+ "loss": 30.0537,
489
+ "step": 325
490
+ },
491
+ {
492
+ "epoch": 0.22,
493
+ "grad_norm": 26.0,
494
+ "learning_rate": 2e-05,
495
+ "loss": 30.3012,
496
+ "step": 330
497
+ },
498
+ {
499
+ "epoch": 0.22333333333333333,
500
+ "grad_norm": 24.0,
501
+ "learning_rate": 2e-05,
502
+ "loss": 29.9831,
503
+ "step": 335
504
+ },
505
+ {
506
+ "epoch": 0.22666666666666666,
507
+ "grad_norm": 5.53125,
508
+ "learning_rate": 2e-05,
509
+ "loss": 30.1414,
510
+ "step": 340
511
+ },
512
+ {
513
+ "epoch": 0.23,
514
+ "grad_norm": 8.5625,
515
+ "learning_rate": 2e-05,
516
+ "loss": 29.8688,
517
+ "step": 345
518
+ },
519
+ {
520
+ "epoch": 0.23333333333333334,
521
+ "grad_norm": 15.375,
522
+ "learning_rate": 2e-05,
523
+ "loss": 29.9341,
524
+ "step": 350
525
+ },
526
+ {
527
+ "epoch": 0.23666666666666666,
528
+ "grad_norm": 4.625,
529
+ "learning_rate": 2e-05,
530
+ "loss": 29.8085,
531
+ "step": 355
532
+ },
533
+ {
534
+ "epoch": 0.24,
535
+ "grad_norm": 5.0625,
536
+ "learning_rate": 2e-05,
537
+ "loss": 29.7093,
538
+ "step": 360
539
+ },
540
+ {
541
+ "epoch": 0.24333333333333335,
542
+ "grad_norm": 6.8125,
543
+ "learning_rate": 2e-05,
544
+ "loss": 29.6956,
545
+ "step": 365
546
+ },
547
+ {
548
+ "epoch": 0.24666666666666667,
549
+ "grad_norm": 5.09375,
550
+ "learning_rate": 2e-05,
551
+ "loss": 29.7583,
552
+ "step": 370
553
+ },
554
+ {
555
+ "epoch": 0.25,
556
+ "grad_norm": 12.625,
557
+ "learning_rate": 2e-05,
558
+ "loss": 29.6723,
559
+ "step": 375
560
+ },
561
+ {
562
+ "epoch": 0.25333333333333335,
563
+ "grad_norm": 9.5,
564
+ "learning_rate": 2e-05,
565
+ "loss": 29.6157,
566
+ "step": 380
567
+ },
568
+ {
569
+ "epoch": 0.25666666666666665,
570
+ "grad_norm": 23.5,
571
+ "learning_rate": 2e-05,
572
+ "loss": 29.6261,
573
+ "step": 385
574
+ },
575
+ {
576
+ "epoch": 0.26,
577
+ "grad_norm": 21.75,
578
+ "learning_rate": 2e-05,
579
+ "loss": 29.6411,
580
+ "step": 390
581
+ },
582
+ {
583
+ "epoch": 0.2633333333333333,
584
+ "grad_norm": 30.25,
585
+ "learning_rate": 2e-05,
586
+ "loss": 29.7162,
587
+ "step": 395
588
+ },
589
+ {
590
+ "epoch": 0.26666666666666666,
591
+ "grad_norm": 8.3125,
592
+ "learning_rate": 2e-05,
593
+ "loss": 29.6952,
594
+ "step": 400
595
+ },
596
+ {
597
+ "epoch": 0.26666666666666666,
598
+ "eval_loss": 7.468276023864746,
599
+ "eval_runtime": 6.838,
600
+ "eval_samples_per_second": 39.339,
601
+ "eval_steps_per_second": 1.024,
602
+ "step": 400
603
  }
604
  ],
605
  "logging_steps": 5,
 
619
  "attributes": {}
620
  }
621
  },
622
+ "total_flos": 2.6162939312472064e+17,
623
  "train_batch_size": 10,
624
  "trial_name": null,
625
  "trial_params": null