CodeIsAbstract commited on
Commit
22db9db
·
verified ·
1 Parent(s): c784d5a

Training in progress, step 4000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:897a290be0c0b16790ad7b090d09b56b33c5889810a7ea5907d4e0fed1bd51ab
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6150dc5bd051bed762db3300592d4cc7ebe9703885d3d6313370582e89e126b1
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ba7808784e33b510417a0442d2514a5dc45d8fa9986862d18084d909efcb30a7
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ced7feb6339932dd38c7dd1b41e44f03af228749aaf328c571f8b3d5e79694bc
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a9fa5fc86e7c65bf8b2609ce132c422acd30ff79c1395383385a52bfa9deec0
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df4b1a89b85b3ec35f63283289082575fe9a15b34a21a7f947d7912552e363be
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3e1c5dfea78108b1145382b3ab5baa3a9f481a062e702dff052035bc19b0def5
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c3220e96802ad35e508b642aabfd7e1cb6a8b7c1925ae81918e4ab19428a5638
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.7,
6
  "eval_steps": 100,
7
- "global_step": 3500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2738,6 +2738,396 @@
2738
  "eval_samples_per_second": 67.472,
2739
  "eval_steps_per_second": 3.849,
2740
  "step": 3500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2741
  }
2742
  ],
2743
  "logging_steps": 10,
@@ -2757,7 +3147,7 @@
2757
  "attributes": {}
2758
  }
2759
  },
2760
- "total_flos": 1.141540802002944e+18,
2761
  "train_batch_size": 18,
2762
  "trial_name": null,
2763
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.8,
6
  "eval_steps": 100,
7
+ "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2738
  "eval_samples_per_second": 67.472,
2739
  "eval_steps_per_second": 3.849,
2740
  "step": 3500
2741
+ },
2742
+ {
2743
+ "epoch": 0.702,
2744
+ "grad_norm": 0.053466796875,
2745
+ "learning_rate": 0.0003,
2746
+ "loss": 2.599693489074707,
2747
+ "step": 3510
2748
+ },
2749
+ {
2750
+ "epoch": 0.704,
2751
+ "grad_norm": 0.07275390625,
2752
+ "learning_rate": 0.0003,
2753
+ "loss": 2.599070167541504,
2754
+ "step": 3520
2755
+ },
2756
+ {
2757
+ "epoch": 0.706,
2758
+ "grad_norm": 0.055419921875,
2759
+ "learning_rate": 0.0003,
2760
+ "loss": 2.6012365341186525,
2761
+ "step": 3530
2762
+ },
2763
+ {
2764
+ "epoch": 0.708,
2765
+ "grad_norm": 0.05224609375,
2766
+ "learning_rate": 0.0003,
2767
+ "loss": 2.576295852661133,
2768
+ "step": 3540
2769
+ },
2770
+ {
2771
+ "epoch": 0.71,
2772
+ "grad_norm": 0.0517578125,
2773
+ "learning_rate": 0.0003,
2774
+ "loss": 2.6024765014648437,
2775
+ "step": 3550
2776
+ },
2777
+ {
2778
+ "epoch": 0.712,
2779
+ "grad_norm": 0.0693359375,
2780
+ "learning_rate": 0.0003,
2781
+ "loss": 2.5685857772827148,
2782
+ "step": 3560
2783
+ },
2784
+ {
2785
+ "epoch": 0.714,
2786
+ "grad_norm": 0.1220703125,
2787
+ "learning_rate": 0.0003,
2788
+ "loss": 2.584480094909668,
2789
+ "step": 3570
2790
+ },
2791
+ {
2792
+ "epoch": 0.716,
2793
+ "grad_norm": 0.06396484375,
2794
+ "learning_rate": 0.0003,
2795
+ "loss": 2.6187564849853517,
2796
+ "step": 3580
2797
+ },
2798
+ {
2799
+ "epoch": 0.718,
2800
+ "grad_norm": 0.05126953125,
2801
+ "learning_rate": 0.0003,
2802
+ "loss": 2.608148384094238,
2803
+ "step": 3590
2804
+ },
2805
+ {
2806
+ "epoch": 0.72,
2807
+ "grad_norm": 0.056396484375,
2808
+ "learning_rate": 0.0003,
2809
+ "loss": 2.596156883239746,
2810
+ "step": 3600
2811
+ },
2812
+ {
2813
+ "epoch": 0.72,
2814
+ "eval_loss": 3.0285751819610596,
2815
+ "eval_runtime": 4.4537,
2816
+ "eval_samples_per_second": 66.91,
2817
+ "eval_steps_per_second": 3.817,
2818
+ "step": 3600
2819
+ },
2820
+ {
2821
+ "epoch": 0.722,
2822
+ "grad_norm": 2.265625,
2823
+ "learning_rate": 0.0003,
2824
+ "loss": 2.6053913116455076,
2825
+ "step": 3610
2826
+ },
2827
+ {
2828
+ "epoch": 0.724,
2829
+ "grad_norm": 0.1259765625,
2830
+ "learning_rate": 0.0003,
2831
+ "loss": 2.5906614303588866,
2832
+ "step": 3620
2833
+ },
2834
+ {
2835
+ "epoch": 0.726,
2836
+ "grad_norm": 0.058349609375,
2837
+ "learning_rate": 0.0003,
2838
+ "loss": 2.6229223251342773,
2839
+ "step": 3630
2840
+ },
2841
+ {
2842
+ "epoch": 0.728,
2843
+ "grad_norm": 0.054443359375,
2844
+ "learning_rate": 0.0003,
2845
+ "loss": 2.6100414276123045,
2846
+ "step": 3640
2847
+ },
2848
+ {
2849
+ "epoch": 0.73,
2850
+ "grad_norm": 0.05078125,
2851
+ "learning_rate": 0.0003,
2852
+ "loss": 2.5861778259277344,
2853
+ "step": 3650
2854
+ },
2855
+ {
2856
+ "epoch": 0.732,
2857
+ "grad_norm": 0.06884765625,
2858
+ "learning_rate": 0.0003,
2859
+ "loss": 2.6088991165161133,
2860
+ "step": 3660
2861
+ },
2862
+ {
2863
+ "epoch": 0.734,
2864
+ "grad_norm": 0.0546875,
2865
+ "learning_rate": 0.0003,
2866
+ "loss": 2.5957805633544924,
2867
+ "step": 3670
2868
+ },
2869
+ {
2870
+ "epoch": 0.736,
2871
+ "grad_norm": 0.08447265625,
2872
+ "learning_rate": 0.0003,
2873
+ "loss": 2.5968517303466796,
2874
+ "step": 3680
2875
+ },
2876
+ {
2877
+ "epoch": 0.738,
2878
+ "grad_norm": 0.06396484375,
2879
+ "learning_rate": 0.0003,
2880
+ "loss": 2.590867614746094,
2881
+ "step": 3690
2882
+ },
2883
+ {
2884
+ "epoch": 0.74,
2885
+ "grad_norm": 0.057373046875,
2886
+ "learning_rate": 0.0003,
2887
+ "loss": 2.6020263671875,
2888
+ "step": 3700
2889
+ },
2890
+ {
2891
+ "epoch": 0.74,
2892
+ "eval_loss": 3.033440113067627,
2893
+ "eval_runtime": 4.4438,
2894
+ "eval_samples_per_second": 67.059,
2895
+ "eval_steps_per_second": 3.826,
2896
+ "step": 3700
2897
+ },
2898
+ {
2899
+ "epoch": 0.742,
2900
+ "grad_norm": 1.8359375,
2901
+ "learning_rate": 0.0003,
2902
+ "loss": 2.624915885925293,
2903
+ "step": 3710
2904
+ },
2905
+ {
2906
+ "epoch": 0.744,
2907
+ "grad_norm": 0.05322265625,
2908
+ "learning_rate": 0.0003,
2909
+ "loss": 2.6158367156982423,
2910
+ "step": 3720
2911
+ },
2912
+ {
2913
+ "epoch": 0.746,
2914
+ "grad_norm": 0.052978515625,
2915
+ "learning_rate": 0.0003,
2916
+ "loss": 2.5968292236328123,
2917
+ "step": 3730
2918
+ },
2919
+ {
2920
+ "epoch": 0.748,
2921
+ "grad_norm": 0.0849609375,
2922
+ "learning_rate": 0.0003,
2923
+ "loss": 2.61802978515625,
2924
+ "step": 3740
2925
+ },
2926
+ {
2927
+ "epoch": 0.75,
2928
+ "grad_norm": 0.05517578125,
2929
+ "learning_rate": 0.0003,
2930
+ "loss": 2.5647909164428713,
2931
+ "step": 3750
2932
+ },
2933
+ {
2934
+ "epoch": 0.752,
2935
+ "grad_norm": 0.060791015625,
2936
+ "learning_rate": 0.0003,
2937
+ "loss": 2.601895332336426,
2938
+ "step": 3760
2939
+ },
2940
+ {
2941
+ "epoch": 0.754,
2942
+ "grad_norm": 0.0556640625,
2943
+ "learning_rate": 0.0003,
2944
+ "loss": 2.605121612548828,
2945
+ "step": 3770
2946
+ },
2947
+ {
2948
+ "epoch": 0.756,
2949
+ "grad_norm": 0.09716796875,
2950
+ "learning_rate": 0.0003,
2951
+ "loss": 2.6213621139526366,
2952
+ "step": 3780
2953
+ },
2954
+ {
2955
+ "epoch": 0.758,
2956
+ "grad_norm": 0.05078125,
2957
+ "learning_rate": 0.0003,
2958
+ "loss": 2.5848875045776367,
2959
+ "step": 3790
2960
+ },
2961
+ {
2962
+ "epoch": 0.76,
2963
+ "grad_norm": 0.05859375,
2964
+ "learning_rate": 0.0003,
2965
+ "loss": 2.585398483276367,
2966
+ "step": 3800
2967
+ },
2968
+ {
2969
+ "epoch": 0.76,
2970
+ "eval_loss": 3.0310301780700684,
2971
+ "eval_runtime": 4.4352,
2972
+ "eval_samples_per_second": 67.19,
2973
+ "eval_steps_per_second": 3.833,
2974
+ "step": 3800
2975
+ },
2976
+ {
2977
+ "epoch": 0.762,
2978
+ "grad_norm": 0.0830078125,
2979
+ "learning_rate": 0.0003,
2980
+ "loss": 2.5891979217529295,
2981
+ "step": 3810
2982
+ },
2983
+ {
2984
+ "epoch": 0.764,
2985
+ "grad_norm": 0.048828125,
2986
+ "learning_rate": 0.0003,
2987
+ "loss": 2.587088203430176,
2988
+ "step": 3820
2989
+ },
2990
+ {
2991
+ "epoch": 0.766,
2992
+ "grad_norm": 0.056884765625,
2993
+ "learning_rate": 0.0003,
2994
+ "loss": 2.6202329635620116,
2995
+ "step": 3830
2996
+ },
2997
+ {
2998
+ "epoch": 0.768,
2999
+ "grad_norm": 0.058349609375,
3000
+ "learning_rate": 0.0003,
3001
+ "loss": 2.588342475891113,
3002
+ "step": 3840
3003
+ },
3004
+ {
3005
+ "epoch": 0.77,
3006
+ "grad_norm": 0.051513671875,
3007
+ "learning_rate": 0.0003,
3008
+ "loss": 2.570342445373535,
3009
+ "step": 3850
3010
+ },
3011
+ {
3012
+ "epoch": 0.772,
3013
+ "grad_norm": 0.054443359375,
3014
+ "learning_rate": 0.0003,
3015
+ "loss": 2.5928205490112304,
3016
+ "step": 3860
3017
+ },
3018
+ {
3019
+ "epoch": 0.774,
3020
+ "grad_norm": 1.1484375,
3021
+ "learning_rate": 0.0003,
3022
+ "loss": 2.562236785888672,
3023
+ "step": 3870
3024
+ },
3025
+ {
3026
+ "epoch": 0.776,
3027
+ "grad_norm": 0.05322265625,
3028
+ "learning_rate": 0.0003,
3029
+ "loss": 2.5821876525878906,
3030
+ "step": 3880
3031
+ },
3032
+ {
3033
+ "epoch": 0.778,
3034
+ "grad_norm": 0.05029296875,
3035
+ "learning_rate": 0.0003,
3036
+ "loss": 2.583372688293457,
3037
+ "step": 3890
3038
+ },
3039
+ {
3040
+ "epoch": 0.78,
3041
+ "grad_norm": 0.197265625,
3042
+ "learning_rate": 0.0003,
3043
+ "loss": 2.5738046646118162,
3044
+ "step": 3900
3045
+ },
3046
+ {
3047
+ "epoch": 0.78,
3048
+ "eval_loss": 3.029505491256714,
3049
+ "eval_runtime": 4.3771,
3050
+ "eval_samples_per_second": 68.082,
3051
+ "eval_steps_per_second": 3.884,
3052
+ "step": 3900
3053
+ },
3054
+ {
3055
+ "epoch": 0.782,
3056
+ "grad_norm": 0.173828125,
3057
+ "learning_rate": 0.0003,
3058
+ "loss": 2.576494598388672,
3059
+ "step": 3910
3060
+ },
3061
+ {
3062
+ "epoch": 0.784,
3063
+ "grad_norm": 0.380859375,
3064
+ "learning_rate": 0.0003,
3065
+ "loss": 2.5919593811035155,
3066
+ "step": 3920
3067
+ },
3068
+ {
3069
+ "epoch": 0.786,
3070
+ "grad_norm": 0.048095703125,
3071
+ "learning_rate": 0.0003,
3072
+ "loss": 2.621496391296387,
3073
+ "step": 3930
3074
+ },
3075
+ {
3076
+ "epoch": 0.788,
3077
+ "grad_norm": 0.052490234375,
3078
+ "learning_rate": 0.0003,
3079
+ "loss": 2.584731101989746,
3080
+ "step": 3940
3081
+ },
3082
+ {
3083
+ "epoch": 0.79,
3084
+ "grad_norm": 0.0498046875,
3085
+ "learning_rate": 0.0003,
3086
+ "loss": 2.568536567687988,
3087
+ "step": 3950
3088
+ },
3089
+ {
3090
+ "epoch": 0.792,
3091
+ "grad_norm": 0.0615234375,
3092
+ "learning_rate": 0.0003,
3093
+ "loss": 2.590066909790039,
3094
+ "step": 3960
3095
+ },
3096
+ {
3097
+ "epoch": 0.794,
3098
+ "grad_norm": 0.060791015625,
3099
+ "learning_rate": 0.0003,
3100
+ "loss": 2.613643264770508,
3101
+ "step": 3970
3102
+ },
3103
+ {
3104
+ "epoch": 0.796,
3105
+ "grad_norm": 0.052734375,
3106
+ "learning_rate": 0.0003,
3107
+ "loss": 2.614491081237793,
3108
+ "step": 3980
3109
+ },
3110
+ {
3111
+ "epoch": 0.798,
3112
+ "grad_norm": 0.0517578125,
3113
+ "learning_rate": 0.0003,
3114
+ "loss": 2.591320037841797,
3115
+ "step": 3990
3116
+ },
3117
+ {
3118
+ "epoch": 0.8,
3119
+ "grad_norm": 0.04931640625,
3120
+ "learning_rate": 0.0003,
3121
+ "loss": 2.605324554443359,
3122
+ "step": 4000
3123
+ },
3124
+ {
3125
+ "epoch": 0.8,
3126
+ "eval_loss": 3.0305187702178955,
3127
+ "eval_runtime": 4.4657,
3128
+ "eval_samples_per_second": 66.73,
3129
+ "eval_steps_per_second": 3.807,
3130
+ "step": 4000
3131
  }
3132
  ],
3133
  "logging_steps": 10,
 
3147
  "attributes": {}
3148
  }
3149
  },
3150
+ "total_flos": 1.304618059431936e+18,
3151
  "train_batch_size": 18,
3152
  "trial_name": null,
3153
  "trial_params": null