CodeIsAbstract commited on
Commit
5993156
·
verified ·
1 Parent(s): 520dcb4

Training in progress, step 52000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8892045571a3d278c702547003643aa319e2898ce1daf1234d43784f70c642a0
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:279e32c6f744dd4b5ca0a4e0bdac260e9eb02f261d15119109be31afee7016e6
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:361bbad0917e897be4b09c5130ad996566c0cdf177169848320c732f7678395a
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c1a242dc9694cfbb9560ec117c694c45ba481f843b0c63ef925a82069fa1e1ba
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f66e48699b199955664d97c152eda4d498658e8c242bfacea1b8e34bd9e4220b
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8e9bb6fca478bc46cc9aedfa2ff02f10ac93eb3ef848bec3ae25bcbbb5f8b50
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:989a59d2631ca368f80e57fef1957ec417fc83d6e6ffb3385f9e62ca59210c21
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:71595add9463a6fb13f2baa718469da29cdbbd86c903fff1dc29e53d6a8513e1
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6233766233766234,
6
  "eval_steps": 1000,
7
- "global_step": 48000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3752,6 +3752,318 @@
3752
  "eval_samples_per_second": 36.675,
3753
  "eval_steps_per_second": 9.169,
3754
  "step": 48000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3755
  }
3756
  ],
3757
  "logging_steps": 100,
@@ -3771,7 +4083,7 @@
3771
  "attributes": {}
3772
  }
3773
  },
3774
- "total_flos": 1.42892706299904e+18,
3775
  "train_batch_size": 22,
3776
  "trial_name": null,
3777
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6753246753246753,
6
  "eval_steps": 1000,
7
+ "global_step": 52000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3752
  "eval_samples_per_second": 36.675,
3753
  "eval_steps_per_second": 9.169,
3754
  "step": 48000
3755
+ },
3756
+ {
3757
+ "epoch": 0.6246753246753247,
3758
+ "grad_norm": 0.2436034381389618,
3759
+ "learning_rate": 0.0006062487050017348,
3760
+ "loss": 2.7583,
3761
+ "step": 48100
3762
+ },
3763
+ {
3764
+ "epoch": 0.625974025974026,
3765
+ "grad_norm": 0.25060826539993286,
3766
+ "learning_rate": 0.0006048394337614477,
3767
+ "loss": 2.7889,
3768
+ "step": 48200
3769
+ },
3770
+ {
3771
+ "epoch": 0.6272727272727273,
3772
+ "grad_norm": 0.23102222383022308,
3773
+ "learning_rate": 0.0006034292908159502,
3774
+ "loss": 2.777,
3775
+ "step": 48300
3776
+ },
3777
+ {
3778
+ "epoch": 0.6285714285714286,
3779
+ "grad_norm": 0.2651401460170746,
3780
+ "learning_rate": 0.000602018287890114,
3781
+ "loss": 2.753,
3782
+ "step": 48400
3783
+ },
3784
+ {
3785
+ "epoch": 0.6298701298701299,
3786
+ "grad_norm": 0.2365243136882782,
3787
+ "learning_rate": 0.000600606436715962,
3788
+ "loss": 2.7541,
3789
+ "step": 48500
3790
+ },
3791
+ {
3792
+ "epoch": 0.6311688311688312,
3793
+ "grad_norm": 0.2493165135383606,
3794
+ "learning_rate": 0.0005991937490325696,
3795
+ "loss": 2.7529,
3796
+ "step": 48600
3797
+ },
3798
+ {
3799
+ "epoch": 0.6324675324675325,
3800
+ "grad_norm": 0.24916312098503113,
3801
+ "learning_rate": 0.0005977802365859677,
3802
+ "loss": 2.7335,
3803
+ "step": 48700
3804
+ },
3805
+ {
3806
+ "epoch": 0.6337662337662338,
3807
+ "grad_norm": 0.38456228375434875,
3808
+ "learning_rate": 0.0005963659111290444,
3809
+ "loss": 2.7492,
3810
+ "step": 48800
3811
+ },
3812
+ {
3813
+ "epoch": 0.6350649350649351,
3814
+ "grad_norm": 0.23968328535556793,
3815
+ "learning_rate": 0.0005949507844214482,
3816
+ "loss": 2.729,
3817
+ "step": 48900
3818
+ },
3819
+ {
3820
+ "epoch": 0.6363636363636364,
3821
+ "grad_norm": 0.27702993154525757,
3822
+ "learning_rate": 0.0005935348682294894,
3823
+ "loss": 2.7453,
3824
+ "step": 49000
3825
+ },
3826
+ {
3827
+ "epoch": 0.6363636363636364,
3828
+ "eval_loss": 3.0973825454711914,
3829
+ "eval_runtime": 15.2932,
3830
+ "eval_samples_per_second": 37.664,
3831
+ "eval_steps_per_second": 9.416,
3832
+ "step": 49000
3833
+ },
3834
+ {
3835
+ "epoch": 0.6376623376623377,
3836
+ "grad_norm": 0.2538464367389679,
3837
+ "learning_rate": 0.000592118174326043,
3838
+ "loss": 2.7291,
3839
+ "step": 49100
3840
+ },
3841
+ {
3842
+ "epoch": 0.638961038961039,
3843
+ "grad_norm": 0.27856433391571045,
3844
+ "learning_rate": 0.0005907007144904501,
3845
+ "loss": 2.7324,
3846
+ "step": 49200
3847
+ },
3848
+ {
3849
+ "epoch": 0.6402597402597403,
3850
+ "grad_norm": 0.26114940643310547,
3851
+ "learning_rate": 0.0005892825005084202,
3852
+ "loss": 2.7777,
3853
+ "step": 49300
3854
+ },
3855
+ {
3856
+ "epoch": 0.6415584415584416,
3857
+ "grad_norm": 0.25457167625427246,
3858
+ "learning_rate": 0.0005878635441719333,
3859
+ "loss": 2.7273,
3860
+ "step": 49400
3861
+ },
3862
+ {
3863
+ "epoch": 0.6428571428571429,
3864
+ "grad_norm": 0.24550481140613556,
3865
+ "learning_rate": 0.0005864438572791422,
3866
+ "loss": 2.7641,
3867
+ "step": 49500
3868
+ },
3869
+ {
3870
+ "epoch": 0.6441558441558441,
3871
+ "grad_norm": 0.2586402893066406,
3872
+ "learning_rate": 0.0005850234516342739,
3873
+ "loss": 2.7115,
3874
+ "step": 49600
3875
+ },
3876
+ {
3877
+ "epoch": 0.6454545454545455,
3878
+ "grad_norm": 0.25716432929039,
3879
+ "learning_rate": 0.000583602339047531,
3880
+ "loss": 2.7164,
3881
+ "step": 49700
3882
+ },
3883
+ {
3884
+ "epoch": 0.6467532467532467,
3885
+ "grad_norm": 0.26527953147888184,
3886
+ "learning_rate": 0.0005821805313349947,
3887
+ "loss": 2.752,
3888
+ "step": 49800
3889
+ },
3890
+ {
3891
+ "epoch": 0.6480519480519481,
3892
+ "grad_norm": 0.24985221028327942,
3893
+ "learning_rate": 0.000580758040318526,
3894
+ "loss": 2.7468,
3895
+ "step": 49900
3896
+ },
3897
+ {
3898
+ "epoch": 0.6493506493506493,
3899
+ "grad_norm": 0.23009921610355377,
3900
+ "learning_rate": 0.0005793348778256671,
3901
+ "loss": 2.7323,
3902
+ "step": 50000
3903
+ },
3904
+ {
3905
+ "epoch": 0.6493506493506493,
3906
+ "eval_loss": 3.0897603034973145,
3907
+ "eval_runtime": 14.9478,
3908
+ "eval_samples_per_second": 38.534,
3909
+ "eval_steps_per_second": 9.634,
3910
+ "step": 50000
3911
+ },
3912
+ {
3913
+ "epoch": 0.6506493506493507,
3914
+ "grad_norm": 0.227068230509758,
3915
+ "learning_rate": 0.0005779110556895433,
3916
+ "loss": 2.7248,
3917
+ "step": 50100
3918
+ },
3919
+ {
3920
+ "epoch": 0.6519480519480519,
3921
+ "grad_norm": 0.22547580301761627,
3922
+ "learning_rate": 0.0005764865857487646,
3923
+ "loss": 2.7372,
3924
+ "step": 50200
3925
+ },
3926
+ {
3927
+ "epoch": 0.6532467532467533,
3928
+ "grad_norm": 0.2620651423931122,
3929
+ "learning_rate": 0.0005750614798473273,
3930
+ "loss": 2.7249,
3931
+ "step": 50300
3932
+ },
3933
+ {
3934
+ "epoch": 0.6545454545454545,
3935
+ "grad_norm": 0.26259052753448486,
3936
+ "learning_rate": 0.0005736357498345156,
3937
+ "loss": 2.6864,
3938
+ "step": 50400
3939
+ },
3940
+ {
3941
+ "epoch": 0.6558441558441559,
3942
+ "grad_norm": 0.24948278069496155,
3943
+ "learning_rate": 0.0005722094075648032,
3944
+ "loss": 2.7276,
3945
+ "step": 50500
3946
+ },
3947
+ {
3948
+ "epoch": 0.6571428571428571,
3949
+ "grad_norm": 0.23260965943336487,
3950
+ "learning_rate": 0.0005707824648977536,
3951
+ "loss": 2.7114,
3952
+ "step": 50600
3953
+ },
3954
+ {
3955
+ "epoch": 0.6584415584415585,
3956
+ "grad_norm": 0.253491073846817,
3957
+ "learning_rate": 0.0005693549336979236,
3958
+ "loss": 2.7156,
3959
+ "step": 50700
3960
+ },
3961
+ {
3962
+ "epoch": 0.6597402597402597,
3963
+ "grad_norm": 0.24990208446979523,
3964
+ "learning_rate": 0.0005679268258347626,
3965
+ "loss": 2.7319,
3966
+ "step": 50800
3967
+ },
3968
+ {
3969
+ "epoch": 0.6610389610389611,
3970
+ "grad_norm": 0.24773485958576202,
3971
+ "learning_rate": 0.0005664981531825152,
3972
+ "loss": 2.7356,
3973
+ "step": 50900
3974
+ },
3975
+ {
3976
+ "epoch": 0.6623376623376623,
3977
+ "grad_norm": 0.23747920989990234,
3978
+ "learning_rate": 0.0005650689276201219,
3979
+ "loss": 2.7497,
3980
+ "step": 51000
3981
+ },
3982
+ {
3983
+ "epoch": 0.6623376623376623,
3984
+ "eval_loss": 3.0903031826019287,
3985
+ "eval_runtime": 14.4116,
3986
+ "eval_samples_per_second": 39.968,
3987
+ "eval_steps_per_second": 9.992,
3988
+ "step": 51000
3989
+ },
3990
+ {
3991
+ "epoch": 0.6636363636363637,
3992
+ "grad_norm": 0.24378643929958344,
3993
+ "learning_rate": 0.0005636391610311204,
3994
+ "loss": 2.7239,
3995
+ "step": 51100
3996
+ },
3997
+ {
3998
+ "epoch": 0.6649350649350649,
3999
+ "grad_norm": 0.2741662561893463,
4000
+ "learning_rate": 0.0005622088653035469,
4001
+ "loss": 2.7161,
4002
+ "step": 51200
4003
+ },
4004
+ {
4005
+ "epoch": 0.6662337662337663,
4006
+ "grad_norm": 0.25329113006591797,
4007
+ "learning_rate": 0.0005607780523298372,
4008
+ "loss": 2.7301,
4009
+ "step": 51300
4010
+ },
4011
+ {
4012
+ "epoch": 0.6675324675324675,
4013
+ "grad_norm": 0.2747024893760681,
4014
+ "learning_rate": 0.0005593467340067282,
4015
+ "loss": 2.7312,
4016
+ "step": 51400
4017
+ },
4018
+ {
4019
+ "epoch": 0.6688311688311688,
4020
+ "grad_norm": 0.24558883905410767,
4021
+ "learning_rate": 0.0005579149222351577,
4022
+ "loss": 2.75,
4023
+ "step": 51500
4024
+ },
4025
+ {
4026
+ "epoch": 0.6701298701298701,
4027
+ "grad_norm": 0.24378612637519836,
4028
+ "learning_rate": 0.0005564826289201674,
4029
+ "loss": 2.742,
4030
+ "step": 51600
4031
+ },
4032
+ {
4033
+ "epoch": 0.6714285714285714,
4034
+ "grad_norm": 0.24434533715248108,
4035
+ "learning_rate": 0.0005550498659708022,
4036
+ "loss": 2.7085,
4037
+ "step": 51700
4038
+ },
4039
+ {
4040
+ "epoch": 0.6727272727272727,
4041
+ "grad_norm": 0.26504042744636536,
4042
+ "learning_rate": 0.000553616645300012,
4043
+ "loss": 2.7147,
4044
+ "step": 51800
4045
+ },
4046
+ {
4047
+ "epoch": 0.674025974025974,
4048
+ "grad_norm": 0.24731750786304474,
4049
+ "learning_rate": 0.0005521829788245528,
4050
+ "loss": 2.7433,
4051
+ "step": 51900
4052
+ },
4053
+ {
4054
+ "epoch": 0.6753246753246753,
4055
+ "grad_norm": 0.23254041373729706,
4056
+ "learning_rate": 0.0005507488784648869,
4057
+ "loss": 2.7326,
4058
+ "step": 52000
4059
+ },
4060
+ {
4061
+ "epoch": 0.6753246753246753,
4062
+ "eval_loss": 3.080451011657715,
4063
+ "eval_runtime": 18.0502,
4064
+ "eval_samples_per_second": 31.911,
4065
+ "eval_steps_per_second": 7.978,
4066
+ "step": 52000
4067
  }
4068
  ],
4069
  "logging_steps": 100,
 
4083
  "attributes": {}
4084
  }
4085
  },
4086
+ "total_flos": 1.54800431824896e+18,
4087
  "train_batch_size": 22,
4088
  "trial_name": null,
4089
  "trial_params": null