CodeIsAbstract commited on
Commit
4955e36
·
verified ·
1 Parent(s): 40248b1

Training in progress, step 400, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:561331e81c6fcc3ea4bd80e90aeeb451e6195fe512d776202f16932e0c17ea6e
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f421426d14674c7d4ad546e19a4e57083d7bca462ab2e13d58cc41736dd8c7fe
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e830fa46b8c9ec45ddf2b32b75587588f6e3d46a65f46e08aad400084a9a6d5
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ccea549a034e18c4f5ecfe1285a4528b857a0949f057eddfb3c088844cb9c36
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9842a5fdf27ba02b590d63e6f1d65edf706e29451025042d306b57ac70d681d7
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca0548a1b90eaf6590b9f6afac3530aaed1157198c5a087c4229ad15001933d9
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:62e2260209115d04c8ea31c913d8d38246600d352056e2ec48e197265ec64063
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d251671b590d29a4724daa893ce2020a6083e1de3b017ea6154c342d8e856b9
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,175 +2,331 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.004,
6
  "eval_steps": 100,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
  "epoch": 0.0002,
14
- "grad_norm": 2672.0,
15
  "learning_rate": 6e-05,
16
- "loss": 26.674310302734376,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.0004,
21
- "grad_norm": 652.0,
22
  "learning_rate": 0.0001266666666666667,
23
- "loss": 26.47657165527344,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.0006,
28
- "grad_norm": 8256.0,
29
  "learning_rate": 0.00019333333333333333,
30
- "loss": 26.448770141601564,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.0008,
35
- "grad_norm": 13632.0,
36
  "learning_rate": 0.00026000000000000003,
37
- "loss": 25.869580078125,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.001,
42
- "grad_norm": 576.0,
43
  "learning_rate": 0.0003266666666666667,
44
- "loss": 25.283935546875,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.0012,
49
- "grad_norm": 212.0,
50
  "learning_rate": 0.0003933333333333333,
51
- "loss": 25.18727264404297,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.0014,
56
- "grad_norm": 109.5,
57
  "learning_rate": 0.00046,
58
- "loss": 24.912762451171876,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.0016,
63
- "grad_norm": 195.0,
64
  "learning_rate": 0.0005266666666666666,
65
- "loss": 24.746240234375,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.0018,
70
- "grad_norm": 1032.0,
71
  "learning_rate": 0.0005933333333333334,
72
- "loss": 24.107341003417968,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.002,
77
- "grad_norm": 135.0,
78
  "learning_rate": 0.00066,
79
- "loss": 23.6096923828125,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.002,
84
- "eval_loss": 3.206618070602417,
85
- "eval_runtime": 4.5304,
86
- "eval_samples_per_second": 65.777,
87
- "eval_steps_per_second": 3.752,
88
  "step": 100
89
  },
90
  {
91
  "epoch": 0.0022,
92
- "grad_norm": 55.75,
93
  "learning_rate": 0.0007266666666666667,
94
- "loss": 22.759126281738283,
95
  "step": 110
96
  },
97
  {
98
  "epoch": 0.0024,
99
- "grad_norm": 8.3125,
100
  "learning_rate": 0.0007933333333333334,
101
- "loss": 22.43407287597656,
102
  "step": 120
103
  },
104
  {
105
  "epoch": 0.0026,
106
- "grad_norm": 2.15625,
107
  "learning_rate": 0.00086,
108
- "loss": 21.722286987304688,
109
  "step": 130
110
  },
111
  {
112
  "epoch": 0.0028,
113
- "grad_norm": 1.046875,
114
  "learning_rate": 0.0009266666666666667,
115
- "loss": 21.480882263183595,
116
  "step": 140
117
  },
118
  {
119
  "epoch": 0.003,
120
- "grad_norm": 1.1875,
121
  "learning_rate": 0.0009933333333333333,
122
- "loss": 21.57379913330078,
123
  "step": 150
124
  },
125
  {
126
  "epoch": 0.0032,
127
- "grad_norm": 0.86328125,
128
  "learning_rate": 0.001,
129
- "loss": 21.262675476074218,
130
  "step": 160
131
  },
132
  {
133
  "epoch": 0.0034,
134
- "grad_norm": 0.703125,
135
  "learning_rate": 0.001,
136
- "loss": 21.385879516601562,
137
  "step": 170
138
  },
139
  {
140
  "epoch": 0.0036,
141
- "grad_norm": 0.84765625,
142
  "learning_rate": 0.001,
143
- "loss": 21.12591552734375,
144
  "step": 180
145
  },
146
  {
147
  "epoch": 0.0038,
148
- "grad_norm": 0.78125,
149
  "learning_rate": 0.001,
150
- "loss": 21.397132873535156,
151
  "step": 190
152
  },
153
  {
154
  "epoch": 0.004,
155
- "grad_norm": 0.67578125,
156
  "learning_rate": 0.001,
157
- "loss": 21.077099609375,
158
  "step": 200
159
  },
160
  {
161
  "epoch": 0.004,
162
- "eval_loss": 3.0551397800445557,
163
- "eval_runtime": 4.4525,
164
- "eval_samples_per_second": 66.929,
165
- "eval_steps_per_second": 3.818,
166
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
167
  }
168
  ],
169
  "logging_steps": 10,
170
  "max_steps": 50000,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 9223372036854775807,
173
- "save_steps": 200,
174
  "stateful_callbacks": {
175
  "TrainerControl": {
176
  "args": {
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.008,
6
  "eval_steps": 100,
7
+ "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
  "epoch": 0.0002,
14
+ "grad_norm": 23680.0,
15
  "learning_rate": 6e-05,
16
+ "loss": 13.378594970703125,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.0004,
21
+ "grad_norm": 308.0,
22
  "learning_rate": 0.0001266666666666667,
23
+ "loss": 13.249179077148437,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.0006,
28
+ "grad_norm": 1240.0,
29
  "learning_rate": 0.00019333333333333333,
30
+ "loss": 13.185873413085938,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.0008,
35
+ "grad_norm": 54528.0,
36
  "learning_rate": 0.00026000000000000003,
37
+ "loss": 12.9256103515625,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.001,
42
+ "grad_norm": 1120.0,
43
  "learning_rate": 0.0003266666666666667,
44
+ "loss": 12.976683044433594,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.0012,
49
+ "grad_norm": 1012.0,
50
  "learning_rate": 0.0003933333333333333,
51
+ "loss": 13.311146545410157,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.0014,
56
+ "grad_norm": 1192.0,
57
  "learning_rate": 0.00046,
58
+ "loss": 13.249429321289062,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.0016,
63
+ "grad_norm": 560.0,
64
  "learning_rate": 0.0005266666666666666,
65
+ "loss": 13.768452453613282,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.0018,
70
+ "grad_norm": 764.0,
71
  "learning_rate": 0.0005933333333333334,
72
+ "loss": 14.640971374511718,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.002,
77
+ "grad_norm": 2736.0,
78
  "learning_rate": 0.00066,
79
+ "loss": 14.4900634765625,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.002,
84
+ "eval_loss": 3.883816957473755,
85
+ "eval_runtime": 4.54,
86
+ "eval_samples_per_second": 65.638,
87
+ "eval_steps_per_second": 3.744,
88
  "step": 100
89
  },
90
  {
91
  "epoch": 0.0022,
92
+ "grad_norm": 5280.0,
93
  "learning_rate": 0.0007266666666666667,
94
+ "loss": 14.381755065917968,
95
  "step": 110
96
  },
97
  {
98
  "epoch": 0.0024,
99
+ "grad_norm": 298.0,
100
  "learning_rate": 0.0007933333333333334,
101
+ "loss": 14.265592956542969,
102
  "step": 120
103
  },
104
  {
105
  "epoch": 0.0026,
106
+ "grad_norm": 45.0,
107
  "learning_rate": 0.00086,
108
+ "loss": 13.5162353515625,
109
  "step": 130
110
  },
111
  {
112
  "epoch": 0.0028,
113
+ "grad_norm": 74.5,
114
  "learning_rate": 0.0009266666666666667,
115
+ "loss": 13.169541931152343,
116
  "step": 140
117
  },
118
  {
119
  "epoch": 0.003,
120
+ "grad_norm": 8.125,
121
  "learning_rate": 0.0009933333333333333,
122
+ "loss": 12.071795654296874,
123
  "step": 150
124
  },
125
  {
126
  "epoch": 0.0032,
127
+ "grad_norm": 2.703125,
128
  "learning_rate": 0.001,
129
+ "loss": 11.409983825683593,
130
  "step": 160
131
  },
132
  {
133
  "epoch": 0.0034,
134
+ "grad_norm": 1.03125,
135
  "learning_rate": 0.001,
136
+ "loss": 10.876200103759766,
137
  "step": 170
138
  },
139
  {
140
  "epoch": 0.0036,
141
+ "grad_norm": 0.91796875,
142
  "learning_rate": 0.001,
143
+ "loss": 10.752876281738281,
144
  "step": 180
145
  },
146
  {
147
  "epoch": 0.0038,
148
+ "grad_norm": 0.71484375,
149
  "learning_rate": 0.001,
150
+ "loss": 10.67746810913086,
151
  "step": 190
152
  },
153
  {
154
  "epoch": 0.004,
155
+ "grad_norm": 5.0,
156
  "learning_rate": 0.001,
157
+ "loss": 10.72914810180664,
158
  "step": 200
159
  },
160
  {
161
  "epoch": 0.004,
162
+ "eval_loss": 3.0593068599700928,
163
+ "eval_runtime": 4.4647,
164
+ "eval_samples_per_second": 66.746,
165
+ "eval_steps_per_second": 3.808,
166
  "step": 200
167
+ },
168
+ {
169
+ "epoch": 0.0042,
170
+ "grad_norm": 12.25,
171
+ "learning_rate": 0.001,
172
+ "loss": 10.68040771484375,
173
+ "step": 210
174
+ },
175
+ {
176
+ "epoch": 0.0044,
177
+ "grad_norm": 0.6640625,
178
+ "learning_rate": 0.001,
179
+ "loss": 10.56131820678711,
180
+ "step": 220
181
+ },
182
+ {
183
+ "epoch": 0.0046,
184
+ "grad_norm": 1.0703125,
185
+ "learning_rate": 0.001,
186
+ "loss": 10.673944854736328,
187
+ "step": 230
188
+ },
189
+ {
190
+ "epoch": 0.0048,
191
+ "grad_norm": 0.59375,
192
+ "learning_rate": 0.001,
193
+ "loss": 10.737371826171875,
194
+ "step": 240
195
+ },
196
+ {
197
+ "epoch": 0.005,
198
+ "grad_norm": 0.546875,
199
+ "learning_rate": 0.001,
200
+ "loss": 10.560675048828125,
201
+ "step": 250
202
+ },
203
+ {
204
+ "epoch": 0.0052,
205
+ "grad_norm": 0.470703125,
206
+ "learning_rate": 0.001,
207
+ "loss": 10.704891204833984,
208
+ "step": 260
209
+ },
210
+ {
211
+ "epoch": 0.0054,
212
+ "grad_norm": 0.494140625,
213
+ "learning_rate": 0.001,
214
+ "loss": 10.544506072998047,
215
+ "step": 270
216
+ },
217
+ {
218
+ "epoch": 0.0056,
219
+ "grad_norm": 0.421875,
220
+ "learning_rate": 0.001,
221
+ "loss": 10.619854736328126,
222
+ "step": 280
223
+ },
224
+ {
225
+ "epoch": 0.0058,
226
+ "grad_norm": 44.5,
227
+ "learning_rate": 0.001,
228
+ "loss": 10.606621551513673,
229
+ "step": 290
230
+ },
231
+ {
232
+ "epoch": 0.006,
233
+ "grad_norm": 0.400390625,
234
+ "learning_rate": 0.001,
235
+ "loss": 10.764106750488281,
236
+ "step": 300
237
+ },
238
+ {
239
+ "epoch": 0.006,
240
+ "eval_loss": 3.049805164337158,
241
+ "eval_runtime": 4.4123,
242
+ "eval_samples_per_second": 67.538,
243
+ "eval_steps_per_second": 3.853,
244
+ "step": 300
245
+ },
246
+ {
247
+ "epoch": 0.0062,
248
+ "grad_norm": 0.45703125,
249
+ "learning_rate": 0.001,
250
+ "loss": 10.587713623046875,
251
+ "step": 310
252
+ },
253
+ {
254
+ "epoch": 0.0064,
255
+ "grad_norm": 0.50390625,
256
+ "learning_rate": 0.001,
257
+ "loss": 10.587606811523438,
258
+ "step": 320
259
+ },
260
+ {
261
+ "epoch": 0.0066,
262
+ "grad_norm": 0.373046875,
263
+ "learning_rate": 0.001,
264
+ "loss": 10.690779876708984,
265
+ "step": 330
266
+ },
267
+ {
268
+ "epoch": 0.0068,
269
+ "grad_norm": 0.33984375,
270
+ "learning_rate": 0.001,
271
+ "loss": 10.637954711914062,
272
+ "step": 340
273
+ },
274
+ {
275
+ "epoch": 0.007,
276
+ "grad_norm": 0.443359375,
277
+ "learning_rate": 0.001,
278
+ "loss": 10.537116241455077,
279
+ "step": 350
280
+ },
281
+ {
282
+ "epoch": 0.0072,
283
+ "grad_norm": 0.5,
284
+ "learning_rate": 0.001,
285
+ "loss": 10.539633178710938,
286
+ "step": 360
287
+ },
288
+ {
289
+ "epoch": 0.0074,
290
+ "grad_norm": 0.72265625,
291
+ "learning_rate": 0.001,
292
+ "loss": 10.701078033447265,
293
+ "step": 370
294
+ },
295
+ {
296
+ "epoch": 0.0076,
297
+ "grad_norm": 0.36328125,
298
+ "learning_rate": 0.001,
299
+ "loss": 10.648773193359375,
300
+ "step": 380
301
+ },
302
+ {
303
+ "epoch": 0.0078,
304
+ "grad_norm": 0.380859375,
305
+ "learning_rate": 0.001,
306
+ "loss": 10.477677154541016,
307
+ "step": 390
308
+ },
309
+ {
310
+ "epoch": 0.008,
311
+ "grad_norm": 7.6875,
312
+ "learning_rate": 0.001,
313
+ "loss": 10.57955093383789,
314
+ "step": 400
315
+ },
316
+ {
317
+ "epoch": 0.008,
318
+ "eval_loss": 3.0504510402679443,
319
+ "eval_runtime": 4.4431,
320
+ "eval_samples_per_second": 67.07,
321
+ "eval_steps_per_second": 3.826,
322
+ "step": 400
323
  }
324
  ],
325
  "logging_steps": 10,
326
  "max_steps": 50000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 9223372036854775807,
329
+ "save_steps": 400,
330
  "stateful_callbacks": {
331
  "TrainerControl": {
332
  "args": {
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:668a0e377ebad57c5d1965a2d2cc09bbdbb8e4a54e766f286eae108d1e754ff9
3
  size 5265
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:666419280b80b3d3942f48be91e30de30a6906e1b7b476ee09f4d3f438f4356a
3
  size 5265