CodeIsAbstract commited on
Commit
b3f4019
·
verified ·
1 Parent(s): 61502e5

Training in progress, step 48000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0302f359737a0861a08d5300ffbacaf83fa3a8be231fe4f4d1d48ec4bc868703
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8892045571a3d278c702547003643aa319e2898ce1daf1234d43784f70c642a0
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c0c65dd0b573072496bd5c019f7a3adc94aeef2c65e8f10a440ac060ae875a09
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:361bbad0917e897be4b09c5130ad996566c0cdf177169848320c732f7678395a
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2ff17261e7b5e5b975d26b5586b146fae1f6df93e96d395e1220068624854c3c
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f66e48699b199955664d97c152eda4d498658e8c242bfacea1b8e34bd9e4220b
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:52822ae525808556d105455f8693d96c34e031a74ca2ba13cffb17b581b87cf5
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:989a59d2631ca368f80e57fef1957ec417fc83d6e6ffb3385f9e62ca59210c21
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.5714285714285714,
6
  "eval_steps": 1000,
7
- "global_step": 44000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3440,6 +3440,318 @@
3440
  "eval_samples_per_second": 37.892,
3441
  "eval_steps_per_second": 9.473,
3442
  "step": 44000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3443
  }
3444
  ],
3445
  "logging_steps": 100,
@@ -3459,7 +3771,7 @@
3459
  "attributes": {}
3460
  }
3461
  },
3462
- "total_flos": 1.30984980774912e+18,
3463
  "train_batch_size": 22,
3464
  "trial_name": null,
3465
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6233766233766234,
6
  "eval_steps": 1000,
7
+ "global_step": 48000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3440
  "eval_samples_per_second": 37.892,
3441
  "eval_steps_per_second": 9.473,
3442
  "step": 44000
3443
+ },
3444
+ {
3445
+ "epoch": 0.5727272727272728,
3446
+ "grad_norm": 0.2565126419067383,
3447
+ "learning_rate": 0.0006617711426628536,
3448
+ "loss": 2.7398,
3449
+ "step": 44100
3450
+ },
3451
+ {
3452
+ "epoch": 0.574025974025974,
3453
+ "grad_norm": 0.22110450267791748,
3454
+ "learning_rate": 0.0006604062599708158,
3455
+ "loss": 2.7204,
3456
+ "step": 44200
3457
+ },
3458
+ {
3459
+ "epoch": 0.5753246753246753,
3460
+ "grad_norm": 0.2212057262659073,
3461
+ "learning_rate": 0.0006590400435537894,
3462
+ "loss": 2.7467,
3463
+ "step": 44300
3464
+ },
3465
+ {
3466
+ "epoch": 0.5766233766233766,
3467
+ "grad_norm": 0.24286267161369324,
3468
+ "learning_rate": 0.0006576725047714116,
3469
+ "loss": 2.7193,
3470
+ "step": 44400
3471
+ },
3472
+ {
3473
+ "epoch": 0.577922077922078,
3474
+ "grad_norm": 0.23778927326202393,
3475
+ "learning_rate": 0.0006563036549943152,
3476
+ "loss": 2.7388,
3477
+ "step": 44500
3478
+ },
3479
+ {
3480
+ "epoch": 0.5792207792207792,
3481
+ "grad_norm": 0.22259333729743958,
3482
+ "learning_rate": 0.000654933505604033,
3483
+ "loss": 2.7296,
3484
+ "step": 44600
3485
+ },
3486
+ {
3487
+ "epoch": 0.5805194805194805,
3488
+ "grad_norm": 0.23351649940013885,
3489
+ "learning_rate": 0.0006535620679929045,
3490
+ "loss": 2.753,
3491
+ "step": 44700
3492
+ },
3493
+ {
3494
+ "epoch": 0.5818181818181818,
3495
+ "grad_norm": 0.22059041261672974,
3496
+ "learning_rate": 0.0006521893535639792,
3497
+ "loss": 2.7281,
3498
+ "step": 44800
3499
+ },
3500
+ {
3501
+ "epoch": 0.5831168831168831,
3502
+ "grad_norm": 0.23977316915988922,
3503
+ "learning_rate": 0.0006508153737309235,
3504
+ "loss": 2.7003,
3505
+ "step": 44900
3506
+ },
3507
+ {
3508
+ "epoch": 0.5844155844155844,
3509
+ "grad_norm": 0.25528889894485474,
3510
+ "learning_rate": 0.0006494401399179255,
3511
+ "loss": 2.7206,
3512
+ "step": 45000
3513
+ },
3514
+ {
3515
+ "epoch": 0.5844155844155844,
3516
+ "eval_loss": 3.1150028705596924,
3517
+ "eval_runtime": 16.5286,
3518
+ "eval_samples_per_second": 34.849,
3519
+ "eval_steps_per_second": 8.712,
3520
+ "step": 45000
3521
+ },
3522
+ {
3523
+ "epoch": 0.5857142857142857,
3524
+ "grad_norm": 0.24675515294075012,
3525
+ "learning_rate": 0.0006480636635595993,
3526
+ "loss": 2.7617,
3527
+ "step": 45100
3528
+ },
3529
+ {
3530
+ "epoch": 0.587012987012987,
3531
+ "grad_norm": 0.23121733963489532,
3532
+ "learning_rate": 0.0006466859561008905,
3533
+ "loss": 2.721,
3534
+ "step": 45200
3535
+ },
3536
+ {
3537
+ "epoch": 0.5883116883116883,
3538
+ "grad_norm": 0.24000558257102966,
3539
+ "learning_rate": 0.0006453070289969807,
3540
+ "loss": 2.7091,
3541
+ "step": 45300
3542
+ },
3543
+ {
3544
+ "epoch": 0.5896103896103896,
3545
+ "grad_norm": 0.22867269814014435,
3546
+ "learning_rate": 0.0006439268937131929,
3547
+ "loss": 2.7395,
3548
+ "step": 45400
3549
+ },
3550
+ {
3551
+ "epoch": 0.5909090909090909,
3552
+ "grad_norm": 0.23441274464130402,
3553
+ "learning_rate": 0.0006425455617248952,
3554
+ "loss": 2.7507,
3555
+ "step": 45500
3556
+ },
3557
+ {
3558
+ "epoch": 0.5922077922077922,
3559
+ "grad_norm": 0.22815178334712982,
3560
+ "learning_rate": 0.0006411630445174063,
3561
+ "loss": 2.7257,
3562
+ "step": 45600
3563
+ },
3564
+ {
3565
+ "epoch": 0.5935064935064935,
3566
+ "grad_norm": 0.22590583562850952,
3567
+ "learning_rate": 0.0006397793535858992,
3568
+ "loss": 2.7303,
3569
+ "step": 45700
3570
+ },
3571
+ {
3572
+ "epoch": 0.5948051948051948,
3573
+ "grad_norm": 0.2853323221206665,
3574
+ "learning_rate": 0.0006383945004353064,
3575
+ "loss": 2.7414,
3576
+ "step": 45800
3577
+ },
3578
+ {
3579
+ "epoch": 0.5961038961038961,
3580
+ "grad_norm": 0.2671266794204712,
3581
+ "learning_rate": 0.000637008496580224,
3582
+ "loss": 2.7558,
3583
+ "step": 45900
3584
+ },
3585
+ {
3586
+ "epoch": 0.5974025974025974,
3587
+ "grad_norm": 0.23534780740737915,
3588
+ "learning_rate": 0.0006356213535448151,
3589
+ "loss": 2.7498,
3590
+ "step": 46000
3591
+ },
3592
+ {
3593
+ "epoch": 0.5974025974025974,
3594
+ "eval_loss": 3.1098833084106445,
3595
+ "eval_runtime": 14.8476,
3596
+ "eval_samples_per_second": 38.794,
3597
+ "eval_steps_per_second": 9.699,
3598
+ "step": 46000
3599
+ },
3600
+ {
3601
+ "epoch": 0.5987012987012987,
3602
+ "grad_norm": 0.25292590260505676,
3603
+ "learning_rate": 0.0006342330828627155,
3604
+ "loss": 2.7116,
3605
+ "step": 46100
3606
+ },
3607
+ {
3608
+ "epoch": 0.6,
3609
+ "grad_norm": 0.24926777184009552,
3610
+ "learning_rate": 0.0006328436960769364,
3611
+ "loss": 2.694,
3612
+ "step": 46200
3613
+ },
3614
+ {
3615
+ "epoch": 0.6012987012987013,
3616
+ "grad_norm": 0.22545288503170013,
3617
+ "learning_rate": 0.0006314532047397697,
3618
+ "loss": 2.7382,
3619
+ "step": 46300
3620
+ },
3621
+ {
3622
+ "epoch": 0.6025974025974026,
3623
+ "grad_norm": 0.24442555010318756,
3624
+ "learning_rate": 0.0006300616204126905,
3625
+ "loss": 2.7408,
3626
+ "step": 46400
3627
+ },
3628
+ {
3629
+ "epoch": 0.6038961038961039,
3630
+ "grad_norm": 0.2365717589855194,
3631
+ "learning_rate": 0.0006286689546662625,
3632
+ "loss": 2.7386,
3633
+ "step": 46500
3634
+ },
3635
+ {
3636
+ "epoch": 0.6051948051948052,
3637
+ "grad_norm": 0.26666414737701416,
3638
+ "learning_rate": 0.0006272752190800404,
3639
+ "loss": 2.7408,
3640
+ "step": 46600
3641
+ },
3642
+ {
3643
+ "epoch": 0.6064935064935065,
3644
+ "grad_norm": 0.2522192597389221,
3645
+ "learning_rate": 0.0006258804252424745,
3646
+ "loss": 2.7487,
3647
+ "step": 46700
3648
+ },
3649
+ {
3650
+ "epoch": 0.6077922077922078,
3651
+ "grad_norm": 0.32666802406311035,
3652
+ "learning_rate": 0.0006244845847508144,
3653
+ "loss": 2.7319,
3654
+ "step": 46800
3655
+ },
3656
+ {
3657
+ "epoch": 0.6090909090909091,
3658
+ "grad_norm": 0.2485225796699524,
3659
+ "learning_rate": 0.0006230877092110119,
3660
+ "loss": 2.7767,
3661
+ "step": 46900
3662
+ },
3663
+ {
3664
+ "epoch": 0.6103896103896104,
3665
+ "grad_norm": 0.28847283124923706,
3666
+ "learning_rate": 0.0006216898102376251,
3667
+ "loss": 2.7883,
3668
+ "step": 47000
3669
+ },
3670
+ {
3671
+ "epoch": 0.6103896103896104,
3672
+ "eval_loss": 3.108853340148926,
3673
+ "eval_runtime": 15.0323,
3674
+ "eval_samples_per_second": 38.317,
3675
+ "eval_steps_per_second": 9.579,
3676
+ "step": 47000
3677
+ },
3678
+ {
3679
+ "epoch": 0.6116883116883117,
3680
+ "grad_norm": 0.2481118142604828,
3681
+ "learning_rate": 0.0006202908994537215,
3682
+ "loss": 2.7383,
3683
+ "step": 47100
3684
+ },
3685
+ {
3686
+ "epoch": 0.612987012987013,
3687
+ "grad_norm": 0.2495453953742981,
3688
+ "learning_rate": 0.0006188909884907814,
3689
+ "loss": 2.7935,
3690
+ "step": 47200
3691
+ },
3692
+ {
3693
+ "epoch": 0.6142857142857143,
3694
+ "grad_norm": 0.257658451795578,
3695
+ "learning_rate": 0.0006174900889886015,
3696
+ "loss": 2.7703,
3697
+ "step": 47300
3698
+ },
3699
+ {
3700
+ "epoch": 0.6155844155844156,
3701
+ "grad_norm": 0.23015721142292023,
3702
+ "learning_rate": 0.0006160882125951977,
3703
+ "loss": 2.7421,
3704
+ "step": 47400
3705
+ },
3706
+ {
3707
+ "epoch": 0.6168831168831169,
3708
+ "grad_norm": 0.2232840657234192,
3709
+ "learning_rate": 0.0006146853709667086,
3710
+ "loss": 2.7372,
3711
+ "step": 47500
3712
+ },
3713
+ {
3714
+ "epoch": 0.6181818181818182,
3715
+ "grad_norm": 0.24996255338191986,
3716
+ "learning_rate": 0.0006132815757672979,
3717
+ "loss": 2.7654,
3718
+ "step": 47600
3719
+ },
3720
+ {
3721
+ "epoch": 0.6194805194805195,
3722
+ "grad_norm": 0.2613258361816406,
3723
+ "learning_rate": 0.0006118768386690587,
3724
+ "loss": 2.7143,
3725
+ "step": 47700
3726
+ },
3727
+ {
3728
+ "epoch": 0.6207792207792208,
3729
+ "grad_norm": 0.2545129656791687,
3730
+ "learning_rate": 0.0006104711713519152,
3731
+ "loss": 2.7443,
3732
+ "step": 47800
3733
+ },
3734
+ {
3735
+ "epoch": 0.6220779220779221,
3736
+ "grad_norm": 0.24030067026615143,
3737
+ "learning_rate": 0.000609064585503526,
3738
+ "loss": 2.7386,
3739
+ "step": 47900
3740
+ },
3741
+ {
3742
+ "epoch": 0.6233766233766234,
3743
+ "grad_norm": 0.23413017392158508,
3744
+ "learning_rate": 0.0006076570928191872,
3745
+ "loss": 2.7276,
3746
+ "step": 48000
3747
+ },
3748
+ {
3749
+ "epoch": 0.6233766233766234,
3750
+ "eval_loss": 3.097559690475464,
3751
+ "eval_runtime": 15.7054,
3752
+ "eval_samples_per_second": 36.675,
3753
+ "eval_steps_per_second": 9.169,
3754
+ "step": 48000
3755
  }
3756
  ],
3757
  "logging_steps": 100,
 
3771
  "attributes": {}
3772
  }
3773
  },
3774
+ "total_flos": 1.42892706299904e+18,
3775
  "train_batch_size": 22,
3776
  "trial_name": null,
3777
  "trial_params": null