mariadg commited on
Commit
00a8cd9
·
verified ·
1 Parent(s): af81e44

Training in progress, epoch 3, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3597be7f38ca31c14f1645560a9ad68b700eca2390d2d5d010f4410debe47901
3
  size 498612824
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8ebcbf923a16a1f92ed2029d7b5096bd4ce2d694e354f4e92f71e3611d674f51
3
  size 498612824
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8ad8d52b875e29d393979c3637e814c8ed7a68ca6941a162b2faa9bf43317d1b
3
  size 997348747
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bee11593ba60756f0666257e18353644213d779ad62bd4e06b8dd79bf580ca6d
3
  size 997348747
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2d8f98c42da24b5a5937db7695e7230d70157ebd07836d9e13ea06997d7dfc52
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:29c06fa0b22dce7337b1de1cac236263e50da150764285d3a599be3edb3e8f1e
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d5c4224dc8a078df7a12c542aa90115ed7c8e4f973bf685055c317752c437b7b
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7403015576a2c5b970874147914d396da1299ab187cad89adf9821e8e17a0f42
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dba91d05f1f723fd97370b2f6d92b20cba8a1ad8076164d6d0112d9ad0660db4
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a68b309a5f1895e4cb5a1d0148608a92726a064ed22b342b2d609135d26b0bd5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 2393,
3
- "best_metric": 0.9541757587780202,
4
- "best_model_checkpoint": "./results_roberta_framing/checkpoint-2393",
5
- "epoch": 2.0,
6
  "eval_steps": 500,
7
- "global_step": 4786,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -699,6 +699,355 @@
699
  "eval_samples_per_second": 408.98,
700
  "eval_steps_per_second": 25.596,
701
  "step": 4786
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
702
  }
703
  ],
704
  "logging_steps": 50,
@@ -713,12 +1062,12 @@
713
  "should_evaluate": false,
714
  "should_log": false,
715
  "should_save": true,
716
- "should_training_stop": false
717
  },
718
  "attributes": {}
719
  }
720
  },
721
- "total_flos": 1.35284510223192e+16,
722
  "train_batch_size": 16,
723
  "trial_name": null,
724
  "trial_params": null
 
1
  {
2
+ "best_global_step": 7179,
3
+ "best_metric": 0.957727228351069,
4
+ "best_model_checkpoint": "./results_roberta_framing/checkpoint-7179",
5
+ "epoch": 3.0,
6
  "eval_steps": 500,
7
+ "global_step": 7179,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
699
  "eval_samples_per_second": 408.98,
700
  "eval_steps_per_second": 25.596,
701
  "step": 4786
702
+ },
703
+ {
704
+ "epoch": 2.0058503969912245,
705
+ "grad_norm": 15.84992790222168,
706
+ "learning_rate": 6.630449923387658e-06,
707
+ "loss": 0.12227592468261719,
708
+ "step": 4800
709
+ },
710
+ {
711
+ "epoch": 2.026744671959883,
712
+ "grad_norm": 0.45001888275146484,
713
+ "learning_rate": 6.491154756929935e-06,
714
+ "loss": 0.06678987979888916,
715
+ "step": 4850
716
+ },
717
+ {
718
+ "epoch": 2.0476389469285414,
719
+ "grad_norm": 0.8402959704399109,
720
+ "learning_rate": 6.351859590472211e-06,
721
+ "loss": 0.11400195121765137,
722
+ "step": 4900
723
+ },
724
+ {
725
+ "epoch": 2.0685332218972,
726
+ "grad_norm": 0.4083273410797119,
727
+ "learning_rate": 6.212564424014487e-06,
728
+ "loss": 0.08374635696411133,
729
+ "step": 4950
730
+ },
731
+ {
732
+ "epoch": 2.089427496865859,
733
+ "grad_norm": 0.15326806902885437,
734
+ "learning_rate": 6.073269257556763e-06,
735
+ "loss": 0.10893898010253907,
736
+ "step": 5000
737
+ },
738
+ {
739
+ "epoch": 2.1103217718345175,
740
+ "grad_norm": 11.168054580688477,
741
+ "learning_rate": 5.93397409109904e-06,
742
+ "loss": 0.1383556079864502,
743
+ "step": 5050
744
+ },
745
+ {
746
+ "epoch": 2.131216046803176,
747
+ "grad_norm": 0.11581141501665115,
748
+ "learning_rate": 5.794678924641315e-06,
749
+ "loss": 0.08604293823242187,
750
+ "step": 5100
751
+ },
752
+ {
753
+ "epoch": 2.1521103217718345,
754
+ "grad_norm": 3.59356427192688,
755
+ "learning_rate": 5.655383758183592e-06,
756
+ "loss": 0.09102669715881348,
757
+ "step": 5150
758
+ },
759
+ {
760
+ "epoch": 2.173004596740493,
761
+ "grad_norm": 0.0897458866238594,
762
+ "learning_rate": 5.516088591725868e-06,
763
+ "loss": 0.05512897491455078,
764
+ "step": 5200
765
+ },
766
+ {
767
+ "epoch": 2.193898871709152,
768
+ "grad_norm": 5.144851207733154,
769
+ "learning_rate": 5.3767934252681445e-06,
770
+ "loss": 0.10218334197998047,
771
+ "step": 5250
772
+ },
773
+ {
774
+ "epoch": 2.21479314667781,
775
+ "grad_norm": 0.227946937084198,
776
+ "learning_rate": 5.23749825881042e-06,
777
+ "loss": 0.09305851936340331,
778
+ "step": 5300
779
+ },
780
+ {
781
+ "epoch": 2.235687421646469,
782
+ "grad_norm": 0.26025137305259705,
783
+ "learning_rate": 5.0982030923526956e-06,
784
+ "loss": 0.07569239139556885,
785
+ "step": 5350
786
+ },
787
+ {
788
+ "epoch": 2.2565816966151275,
789
+ "grad_norm": 0.10767011344432831,
790
+ "learning_rate": 4.9589079258949715e-06,
791
+ "loss": 0.04926969051361084,
792
+ "step": 5400
793
+ },
794
+ {
795
+ "epoch": 2.2774759715837862,
796
+ "grad_norm": 0.4100989103317261,
797
+ "learning_rate": 4.8196127594372475e-06,
798
+ "loss": 0.09085386276245117,
799
+ "step": 5450
800
+ },
801
+ {
802
+ "epoch": 2.2983702465524445,
803
+ "grad_norm": 0.06200162693858147,
804
+ "learning_rate": 4.680317592979524e-06,
805
+ "loss": 0.08421978950500489,
806
+ "step": 5500
807
+ },
808
+ {
809
+ "epoch": 2.319264521521103,
810
+ "grad_norm": 2.6880674362182617,
811
+ "learning_rate": 4.5410224265218e-06,
812
+ "loss": 0.09992499351501465,
813
+ "step": 5550
814
+ },
815
+ {
816
+ "epoch": 2.340158796489762,
817
+ "grad_norm": 0.19707849621772766,
818
+ "learning_rate": 4.401727260064076e-06,
819
+ "loss": 0.04249430656433106,
820
+ "step": 5600
821
+ },
822
+ {
823
+ "epoch": 2.3610530714584206,
824
+ "grad_norm": 0.0682622566819191,
825
+ "learning_rate": 4.262432093606352e-06,
826
+ "loss": 0.08731508255004883,
827
+ "step": 5650
828
+ },
829
+ {
830
+ "epoch": 2.381947346427079,
831
+ "grad_norm": 7.576389789581299,
832
+ "learning_rate": 4.123136927148628e-06,
833
+ "loss": 0.05915518283843994,
834
+ "step": 5700
835
+ },
836
+ {
837
+ "epoch": 2.4028416213957375,
838
+ "grad_norm": 5.807775497436523,
839
+ "learning_rate": 3.983841760690904e-06,
840
+ "loss": 0.08766698837280273,
841
+ "step": 5750
842
+ },
843
+ {
844
+ "epoch": 2.4237358963643962,
845
+ "grad_norm": 25.983449935913086,
846
+ "learning_rate": 3.84454659423318e-06,
847
+ "loss": 0.07257685661315919,
848
+ "step": 5800
849
+ },
850
+ {
851
+ "epoch": 2.444630171333055,
852
+ "grad_norm": 0.14547644555568695,
853
+ "learning_rate": 3.7052514277754565e-06,
854
+ "loss": 0.14911593437194826,
855
+ "step": 5850
856
+ },
857
+ {
858
+ "epoch": 2.465524446301713,
859
+ "grad_norm": 0.35611000657081604,
860
+ "learning_rate": 3.5659562613177325e-06,
861
+ "loss": 0.13339613914489745,
862
+ "step": 5900
863
+ },
864
+ {
865
+ "epoch": 2.486418721270372,
866
+ "grad_norm": 0.08110082149505615,
867
+ "learning_rate": 3.426661094860009e-06,
868
+ "loss": 0.08725551605224609,
869
+ "step": 5950
870
+ },
871
+ {
872
+ "epoch": 2.5073129962390306,
873
+ "grad_norm": 0.5092394351959229,
874
+ "learning_rate": 3.2873659284022845e-06,
875
+ "loss": 0.0801594066619873,
876
+ "step": 6000
877
+ },
878
+ {
879
+ "epoch": 2.5282072712076893,
880
+ "grad_norm": 1.5373930931091309,
881
+ "learning_rate": 3.1480707619445604e-06,
882
+ "loss": 0.1282341194152832,
883
+ "step": 6050
884
+ },
885
+ {
886
+ "epoch": 2.5491015461763475,
887
+ "grad_norm": 0.4443158805370331,
888
+ "learning_rate": 3.008775595486837e-06,
889
+ "loss": 0.0929796314239502,
890
+ "step": 6100
891
+ },
892
+ {
893
+ "epoch": 2.5699958211450062,
894
+ "grad_norm": 10.2298002243042,
895
+ "learning_rate": 2.869480429029113e-06,
896
+ "loss": 0.10379323959350586,
897
+ "step": 6150
898
+ },
899
+ {
900
+ "epoch": 2.590890096113665,
901
+ "grad_norm": 0.0766865611076355,
902
+ "learning_rate": 2.730185262571389e-06,
903
+ "loss": 0.0929744815826416,
904
+ "step": 6200
905
+ },
906
+ {
907
+ "epoch": 2.611784371082323,
908
+ "grad_norm": 0.01663370430469513,
909
+ "learning_rate": 2.590890096113665e-06,
910
+ "loss": 0.0744728136062622,
911
+ "step": 6250
912
+ },
913
+ {
914
+ "epoch": 2.632678646050982,
915
+ "grad_norm": 0.07390784472227097,
916
+ "learning_rate": 2.451594929655941e-06,
917
+ "loss": 0.043207130432128905,
918
+ "step": 6300
919
+ },
920
+ {
921
+ "epoch": 2.6535729210196406,
922
+ "grad_norm": 18.12549591064453,
923
+ "learning_rate": 2.312299763198217e-06,
924
+ "loss": 0.09382804870605468,
925
+ "step": 6350
926
+ },
927
+ {
928
+ "epoch": 2.6744671959882993,
929
+ "grad_norm": 14.898078918457031,
930
+ "learning_rate": 2.1730045967404935e-06,
931
+ "loss": 0.11781403541564942,
932
+ "step": 6400
933
+ },
934
+ {
935
+ "epoch": 2.695361470956958,
936
+ "grad_norm": 15.47895336151123,
937
+ "learning_rate": 2.033709430282769e-06,
938
+ "loss": 0.11296468734741211,
939
+ "step": 6450
940
+ },
941
+ {
942
+ "epoch": 2.7162557459256162,
943
+ "grad_norm": 0.3714354932308197,
944
+ "learning_rate": 1.8944142638250454e-06,
945
+ "loss": 0.09556021690368652,
946
+ "step": 6500
947
+ },
948
+ {
949
+ "epoch": 2.737150020894275,
950
+ "grad_norm": 7.036530494689941,
951
+ "learning_rate": 1.7551190973673216e-06,
952
+ "loss": 0.09722694396972656,
953
+ "step": 6550
954
+ },
955
+ {
956
+ "epoch": 2.7580442958629336,
957
+ "grad_norm": 1.7828617095947266,
958
+ "learning_rate": 1.6158239309095974e-06,
959
+ "loss": 0.08951042175292968,
960
+ "step": 6600
961
+ },
962
+ {
963
+ "epoch": 2.778938570831592,
964
+ "grad_norm": 0.2427712231874466,
965
+ "learning_rate": 1.4765287644518736e-06,
966
+ "loss": 0.08594783782958984,
967
+ "step": 6650
968
+ },
969
+ {
970
+ "epoch": 2.7998328458002506,
971
+ "grad_norm": 0.13741722702980042,
972
+ "learning_rate": 1.3372335979941497e-06,
973
+ "loss": 0.09343706130981445,
974
+ "step": 6700
975
+ },
976
+ {
977
+ "epoch": 2.8207271207689093,
978
+ "grad_norm": 26.75634765625,
979
+ "learning_rate": 1.1979384315364257e-06,
980
+ "loss": 0.05327910423278809,
981
+ "step": 6750
982
+ },
983
+ {
984
+ "epoch": 2.841621395737568,
985
+ "grad_norm": 7.751676082611084,
986
+ "learning_rate": 1.058643265078702e-06,
987
+ "loss": 0.0780732774734497,
988
+ "step": 6800
989
+ },
990
+ {
991
+ "epoch": 2.8625156707062267,
992
+ "grad_norm": 0.48609209060668945,
993
+ "learning_rate": 9.19348098620978e-07,
994
+ "loss": 0.10226208686828614,
995
+ "step": 6850
996
+ },
997
+ {
998
+ "epoch": 2.883409945674885,
999
+ "grad_norm": 35.70721435546875,
1000
+ "learning_rate": 7.80052932163254e-07,
1001
+ "loss": 0.09009994506835937,
1002
+ "step": 6900
1003
+ },
1004
+ {
1005
+ "epoch": 2.9043042206435437,
1006
+ "grad_norm": 0.20764575898647308,
1007
+ "learning_rate": 6.407577657055301e-07,
1008
+ "loss": 0.08318242073059082,
1009
+ "step": 6950
1010
+ },
1011
+ {
1012
+ "epoch": 2.9251984956122024,
1013
+ "grad_norm": 0.6534445285797119,
1014
+ "learning_rate": 5.014625992478062e-07,
1015
+ "loss": 0.09149464607238769,
1016
+ "step": 7000
1017
+ },
1018
+ {
1019
+ "epoch": 2.9460927705808606,
1020
+ "grad_norm": 15.274673461914062,
1021
+ "learning_rate": 3.621674327900822e-07,
1022
+ "loss": 0.06574362754821778,
1023
+ "step": 7050
1024
+ },
1025
+ {
1026
+ "epoch": 2.9669870455495193,
1027
+ "grad_norm": 0.4896354675292969,
1028
+ "learning_rate": 2.2287226633235828e-07,
1029
+ "loss": 0.08324499130249023,
1030
+ "step": 7100
1031
+ },
1032
+ {
1033
+ "epoch": 2.987881320518178,
1034
+ "grad_norm": 0.035827573388814926,
1035
+ "learning_rate": 8.357709987463436e-08,
1036
+ "loss": 0.09447690010070801,
1037
+ "step": 7150
1038
+ },
1039
+ {
1040
+ "epoch": 3.0,
1041
+ "eval_accuracy": 0.9547591683209696,
1042
+ "eval_f1": 0.957727228351069,
1043
+ "eval_loss": 0.21484783291816711,
1044
+ "eval_precision": 0.9646017699115044,
1045
+ "eval_recall": 0.9509499806126406,
1046
+ "eval_roc_auc": 0.9897728933171557,
1047
+ "eval_runtime": 23.436,
1048
+ "eval_samples_per_second": 408.389,
1049
+ "eval_steps_per_second": 25.559,
1050
+ "step": 7179
1051
  }
1052
  ],
1053
  "logging_steps": 50,
 
1062
  "should_evaluate": false,
1063
  "should_log": false,
1064
  "should_save": true,
1065
+ "should_training_stop": true
1066
  },
1067
  "attributes": {}
1068
  }
1069
  },
1070
+ "total_flos": 2.025956770716096e+16,
1071
  "train_batch_size": 16,
1072
  "trial_name": null,
1073
  "trial_params": null