CodeIsAbstract commited on
Commit
6da1f4a
·
verified ·
1 Parent(s): 09ff2cd

Training in progress, step 500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9d91f361723cdb3da7122704a5a753827705ee51e8d9db8d95d17fa16e1214a1
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5dd7f230851339b861737ac7f67ae0ed9375ce1b110bdab60cbcb0423e7fa234
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:be60ae141d864e2c961c2d98241d3c95b42ab05195017b53e1dd1fb3455bc2c7
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9b6baba95860d4d92494572f788ea8c87710f9e3ab9a13bcb44e1ff7ec7c834a
3
  size 469516363
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dfc5c6949cf8a9eca4d88710f0c9c06f410fb63454fb60c644727889ad457648
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ba487b3335620f3a6e123aeda5007c1c83d8788fc08709b71729143cfcd3d74
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,409 +2,409 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.05555555555555555,
6
- "eval_steps": 200,
7
- "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0011111111111111111,
14
- "grad_norm": 95.5,
 
 
 
 
 
 
 
15
  "learning_rate": 0.00011399999999999999,
16
- "loss": 3.3263259887695313,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0022222222222222222,
21
- "grad_norm": 37.75,
22
- "learning_rate": 0.000234,
23
- "loss": 3.246612548828125,
24
- "step": 40
25
  },
26
  {
27
- "epoch": 0.0033333333333333335,
28
- "grad_norm": 72.0,
29
- "learning_rate": 0.0003,
30
- "loss": 3.197848892211914,
31
- "step": 60
32
  },
33
  {
34
- "epoch": 0.0044444444444444444,
35
- "grad_norm": 137.0,
36
- "learning_rate": 0.0003,
37
- "loss": 3.0936569213867187,
38
- "step": 80
39
  },
40
  {
41
- "epoch": 0.005555555555555556,
42
- "grad_norm": 13.8125,
43
- "learning_rate": 0.0003,
44
- "loss": 3.0532867431640627,
45
- "step": 100
46
- },
47
- {
48
- "epoch": 0.006666666666666667,
49
- "grad_norm": 11.875,
50
- "learning_rate": 0.0003,
51
- "loss": 3.0499711990356446,
52
- "step": 120
53
  },
54
  {
55
- "epoch": 0.0077777777777777776,
56
- "grad_norm": 34.0,
57
- "learning_rate": 0.0003,
58
- "loss": 2.9780717849731446,
59
- "step": 140
60
  },
61
  {
62
- "epoch": 0.008888888888888889,
63
- "grad_norm": 16.375,
64
- "learning_rate": 0.0003,
65
- "loss": 2.9729740142822267,
66
- "step": 160
67
  },
68
  {
69
- "epoch": 0.01,
70
- "grad_norm": 1.6640625,
71
- "learning_rate": 0.0003,
72
- "loss": 2.905041313171387,
73
- "step": 180
74
  },
75
  {
76
- "epoch": 0.011111111111111112,
77
- "grad_norm": 2.5625,
78
- "learning_rate": 0.0003,
79
- "loss": 2.8482807159423826,
80
- "step": 200
81
  },
82
  {
83
- "epoch": 0.011111111111111112,
84
- "eval_loss": 3.1418681144714355,
85
- "eval_runtime": 4.509,
86
- "eval_samples_per_second": 66.091,
87
- "eval_steps_per_second": 3.77,
88
- "step": 200
89
  },
90
  {
91
- "epoch": 0.012222222222222223,
92
- "grad_norm": 1.6875,
93
- "learning_rate": 0.0003,
94
- "loss": 2.763028144836426,
95
- "step": 220
96
  },
97
  {
98
- "epoch": 0.013333333333333334,
99
- "grad_norm": 0.5078125,
100
- "learning_rate": 0.0003,
101
- "loss": 2.7428956985473634,
102
- "step": 240
103
  },
104
  {
105
- "epoch": 0.014444444444444444,
106
- "grad_norm": 0.2578125,
107
- "learning_rate": 0.0003,
108
- "loss": 2.6967418670654295,
109
- "step": 260
110
  },
111
  {
112
- "epoch": 0.015555555555555555,
113
- "grad_norm": 0.357421875,
114
- "learning_rate": 0.0003,
115
- "loss": 2.666821098327637,
116
- "step": 280
117
  },
118
  {
119
- "epoch": 0.016666666666666666,
120
- "grad_norm": 0.404296875,
121
- "learning_rate": 0.0003,
122
- "loss": 2.6833669662475588,
123
- "step": 300
124
  },
125
  {
126
- "epoch": 0.017777777777777778,
127
- "grad_norm": 0.142578125,
128
- "learning_rate": 0.0003,
129
- "loss": 2.6547067642211912,
130
- "step": 320
131
  },
132
  {
133
- "epoch": 0.01888888888888889,
134
- "grad_norm": 0.10986328125,
135
- "learning_rate": 0.0003,
136
- "loss": 2.6726154327392577,
137
- "step": 340
138
  },
139
  {
140
- "epoch": 0.02,
141
- "grad_norm": 0.310546875,
142
- "learning_rate": 0.0003,
143
- "loss": 2.639314079284668,
144
- "step": 360
145
  },
146
  {
147
- "epoch": 0.021111111111111112,
148
- "grad_norm": 0.265625,
149
- "learning_rate": 0.0003,
150
- "loss": 2.6715606689453124,
151
- "step": 380
152
  },
153
  {
154
- "epoch": 0.022222222222222223,
155
- "grad_norm": 0.15234375,
156
- "learning_rate": 0.0003,
157
- "loss": 2.633011817932129,
158
- "step": 400
159
  },
160
  {
161
- "epoch": 0.022222222222222223,
162
- "eval_loss": 3.0355894565582275,
163
- "eval_runtime": 4.4466,
164
- "eval_samples_per_second": 67.017,
165
- "eval_steps_per_second": 3.823,
166
- "step": 400
167
  },
168
  {
169
- "epoch": 0.023333333333333334,
170
- "grad_norm": 0.9296875,
171
- "learning_rate": 0.0003,
172
- "loss": 2.6349510192871093,
173
- "step": 420
174
  },
175
  {
176
- "epoch": 0.024444444444444446,
177
- "grad_norm": 0.25,
178
- "learning_rate": 0.0003,
179
- "loss": 2.6378190994262694,
180
- "step": 440
181
  },
182
  {
183
- "epoch": 0.025555555555555557,
184
- "grad_norm": 0.091796875,
185
- "learning_rate": 0.0003,
186
- "loss": 2.645473670959473,
187
- "step": 460
188
  },
189
  {
190
- "epoch": 0.02666666666666667,
191
- "grad_norm": 0.7109375,
192
- "learning_rate": 0.0003,
193
- "loss": 2.6475948333740233,
194
- "step": 480
195
  },
196
  {
197
- "epoch": 0.027777777777777776,
198
- "grad_norm": 0.0927734375,
199
- "learning_rate": 0.0003,
200
- "loss": 2.652556228637695,
201
- "step": 500
202
  },
203
  {
204
- "epoch": 0.028888888888888888,
205
- "grad_norm": 0.10400390625,
206
- "learning_rate": 0.0003,
207
- "loss": 2.6353578567504883,
208
- "step": 520
209
  },
210
  {
211
- "epoch": 0.03,
212
- "grad_norm": 0.11767578125,
213
- "learning_rate": 0.0003,
214
- "loss": 2.617463493347168,
215
- "step": 540
216
  },
217
  {
218
- "epoch": 0.03111111111111111,
219
- "grad_norm": 0.09619140625,
220
- "learning_rate": 0.0003,
221
- "loss": 2.6145917892456056,
222
- "step": 560
223
  },
224
  {
225
- "epoch": 0.03222222222222222,
226
- "grad_norm": 0.291015625,
227
- "learning_rate": 0.0003,
228
- "loss": 2.6116378784179686,
229
- "step": 580
230
  },
231
  {
232
- "epoch": 0.03333333333333333,
233
- "grad_norm": 0.09423828125,
234
- "learning_rate": 0.0003,
235
- "loss": 2.6324670791625975,
236
- "step": 600
237
  },
238
  {
239
- "epoch": 0.03333333333333333,
240
- "eval_loss": 3.0325393676757812,
241
- "eval_runtime": 4.4279,
242
- "eval_samples_per_second": 67.301,
243
- "eval_steps_per_second": 3.839,
244
- "step": 600
245
  },
246
  {
247
- "epoch": 0.034444444444444444,
248
- "grad_norm": 0.09912109375,
249
- "learning_rate": 0.0003,
250
- "loss": 2.627296257019043,
251
- "step": 620
252
  },
253
  {
254
- "epoch": 0.035555555555555556,
255
- "grad_norm": 0.1064453125,
256
- "learning_rate": 0.0003,
257
- "loss": 2.615768623352051,
258
- "step": 640
259
  },
260
  {
261
- "epoch": 0.03666666666666667,
262
- "grad_norm": 0.0830078125,
263
- "learning_rate": 0.0003,
264
- "loss": 2.630397415161133,
265
- "step": 660
266
  },
267
  {
268
- "epoch": 0.03777777777777778,
269
- "grad_norm": 2.34375,
270
- "learning_rate": 0.0003,
271
- "loss": 2.6295297622680662,
272
- "step": 680
273
  },
274
  {
275
- "epoch": 0.03888888888888889,
276
- "grad_norm": 0.091796875,
277
- "learning_rate": 0.0003,
278
- "loss": 2.6139799118041993,
279
- "step": 700
280
  },
281
  {
282
- "epoch": 0.04,
283
- "grad_norm": 0.177734375,
284
- "learning_rate": 0.0003,
285
- "loss": 2.612909507751465,
286
- "step": 720
287
  },
288
  {
289
- "epoch": 0.04111111111111111,
290
- "grad_norm": 0.08984375,
291
- "learning_rate": 0.0003,
292
- "loss": 2.637497901916504,
293
- "step": 740
294
  },
295
  {
296
- "epoch": 0.042222222222222223,
297
- "grad_norm": 0.091796875,
298
- "learning_rate": 0.0003,
299
- "loss": 2.6311773300170898,
300
- "step": 760
301
  },
302
  {
303
- "epoch": 0.043333333333333335,
304
- "grad_norm": 0.1748046875,
305
- "learning_rate": 0.0003,
306
- "loss": 2.631211280822754,
307
- "step": 780
308
  },
309
  {
310
- "epoch": 0.044444444444444446,
311
- "grad_norm": 0.08154296875,
312
- "learning_rate": 0.0003,
313
- "loss": 2.64044075012207,
314
- "step": 800
315
  },
316
  {
317
- "epoch": 0.044444444444444446,
318
- "eval_loss": 3.029780149459839,
319
- "eval_runtime": 4.4461,
320
- "eval_samples_per_second": 67.025,
321
- "eval_steps_per_second": 3.824,
322
- "step": 800
323
  },
324
  {
325
- "epoch": 0.04555555555555556,
326
- "grad_norm": 0.0810546875,
327
- "learning_rate": 0.0003,
328
- "loss": 2.6481130599975584,
329
- "step": 820
330
  },
331
  {
332
- "epoch": 0.04666666666666667,
333
- "grad_norm": 0.06884765625,
334
- "learning_rate": 0.0003,
335
- "loss": 2.6261564254760743,
336
- "step": 840
337
  },
338
  {
339
- "epoch": 0.04777777777777778,
340
- "grad_norm": 0.07373046875,
341
- "learning_rate": 0.0003,
342
- "loss": 2.611224365234375,
343
- "step": 860
344
  },
345
  {
346
- "epoch": 0.04888888888888889,
347
- "grad_norm": 0.119140625,
348
- "learning_rate": 0.0003,
349
- "loss": 2.636740875244141,
350
- "step": 880
351
  },
352
  {
353
- "epoch": 0.05,
354
- "grad_norm": 0.06884765625,
355
- "learning_rate": 0.0003,
356
- "loss": 2.614007568359375,
357
- "step": 900
358
  },
359
  {
360
- "epoch": 0.051111111111111114,
361
- "grad_norm": 0.068359375,
362
- "learning_rate": 0.0003,
363
- "loss": 2.616529846191406,
364
- "step": 920
365
  },
366
  {
367
- "epoch": 0.052222222222222225,
368
- "grad_norm": 0.07470703125,
369
- "learning_rate": 0.0003,
370
- "loss": 2.6233530044555664,
371
- "step": 940
372
  },
373
  {
374
- "epoch": 0.05333333333333334,
375
- "grad_norm": 0.06689453125,
376
- "learning_rate": 0.0003,
377
- "loss": 2.600985527038574,
378
- "step": 960
379
  },
380
  {
381
- "epoch": 0.05444444444444444,
382
- "grad_norm": 0.0791015625,
383
- "learning_rate": 0.0003,
384
- "loss": 2.6462072372436523,
385
- "step": 980
386
  },
387
  {
388
- "epoch": 0.05555555555555555,
389
- "grad_norm": 0.07470703125,
390
- "learning_rate": 0.0003,
391
- "loss": 2.6183977127075195,
392
- "step": 1000
393
  },
394
  {
395
- "epoch": 0.05555555555555555,
396
- "eval_loss": 3.0301759243011475,
397
- "eval_runtime": 4.4446,
398
- "eval_samples_per_second": 67.048,
399
- "eval_steps_per_second": 3.825,
400
- "step": 1000
401
  }
402
  ],
403
- "logging_steps": 20,
404
- "max_steps": 18000,
405
  "num_input_tokens_seen": 0,
406
  "num_train_epochs": 9223372036854775807,
407
- "save_steps": 1000,
408
  "stateful_callbacks": {
409
  "TrainerControl": {
410
  "args": {
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.07142857142857142,
6
+ "eval_steps": 100,
7
+ "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0014285714285714286,
14
+ "grad_norm": 160.0,
15
+ "learning_rate": 5.399999999999999e-05,
16
+ "loss": 3.334370803833008,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.002857142857142857,
21
+ "grad_norm": 143.0,
22
  "learning_rate": 0.00011399999999999999,
23
+ "loss": 3.312285232543945,
24
  "step": 20
25
  },
26
  {
27
+ "epoch": 0.004285714285714286,
28
+ "grad_norm": 91.0,
29
+ "learning_rate": 0.00017399999999999997,
30
+ "loss": 3.3183692932128905,
31
+ "step": 30
32
  },
33
  {
34
+ "epoch": 0.005714285714285714,
35
+ "grad_norm": 306.0,
36
+ "learning_rate": 0.000234,
37
+ "loss": 3.232763671875,
38
+ "step": 40
39
  },
40
  {
41
+ "epoch": 0.007142857142857143,
42
+ "grad_norm": 114.5,
43
+ "learning_rate": 0.000294,
44
+ "loss": 3.1770435333251954,
45
+ "step": 50
46
  },
47
  {
48
+ "epoch": 0.008571428571428572,
49
+ "grad_norm": 148.0,
50
+ "learning_rate": 0.00029999875870267496,
51
+ "loss": 3.1800256729125977,
52
+ "step": 60
 
 
 
 
 
 
 
53
  },
54
  {
55
+ "epoch": 0.01,
56
+ "grad_norm": 25.0,
57
+ "learning_rate": 0.0002999944678247173,
58
+ "loss": 3.166558837890625,
59
+ "step": 70
60
  },
61
  {
62
+ "epoch": 0.011428571428571429,
63
+ "grad_norm": 22.75,
64
+ "learning_rate": 0.0002999871121291227,
65
+ "loss": 3.203672409057617,
66
+ "step": 80
67
  },
68
  {
69
+ "epoch": 0.012857142857142857,
70
+ "grad_norm": 6.4375,
71
+ "learning_rate": 0.00029997669176618923,
72
+ "loss": 3.155362129211426,
73
+ "step": 90
74
  },
75
  {
76
+ "epoch": 0.014285714285714285,
77
+ "grad_norm": 142.0,
78
+ "learning_rate": 0.0002999632069488347,
79
+ "loss": 3.1418533325195312,
80
+ "step": 100
81
  },
82
  {
83
+ "epoch": 0.014285714285714285,
84
+ "eval_loss": 3.406630277633667,
85
+ "eval_runtime": 4.5564,
86
+ "eval_samples_per_second": 65.402,
87
+ "eval_steps_per_second": 3.731,
88
+ "step": 100
89
  },
90
  {
91
+ "epoch": 0.015714285714285715,
92
+ "grad_norm": 200.0,
93
+ "learning_rate": 0.00029994665795259274,
94
+ "loss": 3.106761932373047,
95
+ "step": 110
96
  },
97
  {
98
+ "epoch": 0.017142857142857144,
99
+ "grad_norm": 193.0,
100
+ "learning_rate": 0.0002999270451156068,
101
+ "loss": 3.1113397598266603,
102
+ "step": 120
103
  },
104
  {
105
+ "epoch": 0.018571428571428572,
106
+ "grad_norm": 39.5,
107
+ "learning_rate": 0.0002999043688386235,
108
+ "loss": 3.0628725051879884,
109
+ "step": 130
110
  },
111
  {
112
+ "epoch": 0.02,
113
+ "grad_norm": 3.125,
114
+ "learning_rate": 0.0002998786295849843,
115
+ "loss": 3.030979347229004,
116
+ "step": 140
117
  },
118
  {
119
+ "epoch": 0.02142857142857143,
120
+ "grad_norm": 10.0625,
121
+ "learning_rate": 0.000299849827880616,
122
+ "loss": 3.0148197174072267,
123
+ "step": 150
124
  },
125
  {
126
+ "epoch": 0.022857142857142857,
127
+ "grad_norm": 4.03125,
128
+ "learning_rate": 0.00029981796431402015,
129
+ "loss": 2.963352394104004,
130
+ "step": 160
131
  },
132
  {
133
+ "epoch": 0.024285714285714285,
134
+ "grad_norm": 28.5,
135
+ "learning_rate": 0.0002997830395362608,
136
+ "loss": 2.951630401611328,
137
+ "step": 170
138
  },
139
  {
140
+ "epoch": 0.025714285714285714,
141
+ "grad_norm": 5.15625,
142
+ "learning_rate": 0.00029974505426095166,
143
+ "loss": 2.903683662414551,
144
+ "step": 180
145
  },
146
  {
147
+ "epoch": 0.027142857142857142,
148
+ "grad_norm": 2.359375,
149
+ "learning_rate": 0.0002997040092642407,
150
+ "loss": 2.911179542541504,
151
+ "step": 190
152
  },
153
  {
154
+ "epoch": 0.02857142857142857,
155
+ "grad_norm": 1.484375,
156
+ "learning_rate": 0.00029965990538479526,
157
+ "loss": 2.8417181015014648,
158
+ "step": 200
159
  },
160
  {
161
+ "epoch": 0.02857142857142857,
162
+ "eval_loss": 3.159050703048706,
163
+ "eval_runtime": 4.4564,
164
+ "eval_samples_per_second": 66.87,
165
+ "eval_steps_per_second": 3.815,
166
+ "step": 200
167
  },
168
  {
169
+ "epoch": 0.03,
170
+ "grad_norm": 1.515625,
171
+ "learning_rate": 0.0002996127435237841,
172
+ "loss": 2.815742301940918,
173
+ "step": 210
174
  },
175
  {
176
+ "epoch": 0.03142857142857143,
177
+ "grad_norm": 1.71875,
178
+ "learning_rate": 0.0002995625246448595,
179
+ "loss": 2.7929365158081056,
180
+ "step": 220
181
  },
182
  {
183
+ "epoch": 0.032857142857142856,
184
+ "grad_norm": 0.765625,
185
+ "learning_rate": 0.00029950924977413735,
186
+ "loss": 2.7637331008911135,
187
+ "step": 230
188
  },
189
  {
190
+ "epoch": 0.03428571428571429,
191
+ "grad_norm": 0.609375,
192
+ "learning_rate": 0.0002994529200001762,
193
+ "loss": 2.73776912689209,
194
+ "step": 240
195
  },
196
  {
197
+ "epoch": 0.03571428571428571,
198
+ "grad_norm": 1.5703125,
199
+ "learning_rate": 0.00029939353647395506,
200
+ "loss": 2.723506736755371,
201
+ "step": 250
202
  },
203
  {
204
+ "epoch": 0.037142857142857144,
205
+ "grad_norm": 0.66015625,
206
+ "learning_rate": 0.00029933110040884987,
207
+ "loss": 2.6915863037109373,
208
+ "step": 260
209
  },
210
  {
211
+ "epoch": 0.03857142857142857,
212
+ "grad_norm": 0.9609375,
213
+ "learning_rate": 0.00029926561308060874,
214
+ "loss": 2.663774871826172,
215
+ "step": 270
216
  },
217
  {
218
+ "epoch": 0.04,
219
+ "grad_norm": 0.65234375,
220
+ "learning_rate": 0.00029919707582732577,
221
+ "loss": 2.6552204132080077,
222
+ "step": 280
223
  },
224
  {
225
+ "epoch": 0.041428571428571426,
226
+ "grad_norm": 0.1708984375,
227
+ "learning_rate": 0.0002991254900494139,
228
+ "loss": 2.6489795684814452,
229
+ "step": 290
230
  },
231
  {
232
+ "epoch": 0.04285714285714286,
233
+ "grad_norm": 0.984375,
234
+ "learning_rate": 0.000299050857209576,
235
+ "loss": 2.6665163040161133,
236
+ "step": 300
237
  },
238
  {
239
+ "epoch": 0.04285714285714286,
240
+ "eval_loss": 3.0404820442199707,
241
+ "eval_runtime": 4.4652,
242
+ "eval_samples_per_second": 66.739,
243
+ "eval_steps_per_second": 3.807,
244
+ "step": 300
245
  },
246
  {
247
+ "epoch": 0.04428571428571428,
248
+ "grad_norm": 1.2890625,
249
+ "learning_rate": 0.00029897317883277537,
250
+ "loss": 2.6572853088378907,
251
+ "step": 310
252
  },
253
  {
254
+ "epoch": 0.045714285714285714,
255
+ "grad_norm": 0.2265625,
256
+ "learning_rate": 0.00029889245650620413,
257
+ "loss": 2.642633056640625,
258
+ "step": 320
259
  },
260
  {
261
+ "epoch": 0.047142857142857146,
262
+ "grad_norm": 0.6484375,
263
+ "learning_rate": 0.0002988086918792514,
264
+ "loss": 2.6547801971435545,
265
+ "step": 330
266
  },
267
  {
268
+ "epoch": 0.04857142857142857,
269
+ "grad_norm": 4.625,
270
+ "learning_rate": 0.0002987218866634688,
271
+ "loss": 2.651763153076172,
272
+ "step": 340
273
  },
274
  {
275
+ "epoch": 0.05,
276
+ "grad_norm": 0.2275390625,
277
+ "learning_rate": 0.00029863204263253624,
278
+ "loss": 2.635426902770996,
279
+ "step": 350
280
  },
281
  {
282
+ "epoch": 0.05142857142857143,
283
+ "grad_norm": 0.11328125,
284
+ "learning_rate": 0.0002985391616222252,
285
+ "loss": 2.6323591232299806,
286
+ "step": 360
287
  },
288
  {
289
+ "epoch": 0.05285714285714286,
290
+ "grad_norm": 3.125,
291
+ "learning_rate": 0.0002984432455303614,
292
+ "loss": 2.655726432800293,
293
+ "step": 370
294
  },
295
  {
296
+ "epoch": 0.054285714285714284,
297
+ "grad_norm": 0.3359375,
298
+ "learning_rate": 0.00029834429631678597,
299
+ "loss": 2.6482254028320313,
300
+ "step": 380
301
  },
302
  {
303
+ "epoch": 0.055714285714285716,
304
+ "grad_norm": 0.54296875,
305
+ "learning_rate": 0.00029824231600331547,
306
+ "loss": 2.6480077743530273,
307
+ "step": 390
308
  },
309
  {
310
+ "epoch": 0.05714285714285714,
311
+ "grad_norm": 0.166015625,
312
+ "learning_rate": 0.0002981373066737005,
313
+ "loss": 2.6565391540527346,
314
+ "step": 400
315
  },
316
  {
317
+ "epoch": 0.05714285714285714,
318
+ "eval_loss": 3.0312750339508057,
319
+ "eval_runtime": 4.4433,
320
+ "eval_samples_per_second": 67.068,
321
+ "eval_steps_per_second": 3.826,
322
+ "step": 400
323
  },
324
  {
325
+ "epoch": 0.05857142857142857,
326
+ "grad_norm": 0.255859375,
327
+ "learning_rate": 0.0002980292704735831,
328
+ "loss": 2.663307762145996,
329
+ "step": 410
330
  },
331
  {
332
+ "epoch": 0.06,
333
+ "grad_norm": 0.0966796875,
334
+ "learning_rate": 0.00029791820961045317,
335
+ "loss": 2.6410280227661134,
336
+ "step": 420
337
  },
338
  {
339
+ "epoch": 0.06142857142857143,
340
+ "grad_norm": 0.3125,
341
+ "learning_rate": 0.0002978041263536029,
342
+ "loss": 2.6257579803466795,
343
+ "step": 430
344
  },
345
  {
346
+ "epoch": 0.06285714285714286,
347
+ "grad_norm": 0.310546875,
348
+ "learning_rate": 0.0002976870230340808,
349
+ "loss": 2.650602912902832,
350
+ "step": 440
351
  },
352
  {
353
+ "epoch": 0.06428571428571428,
354
+ "grad_norm": 0.115234375,
355
+ "learning_rate": 0.0002975669020446439,
356
+ "loss": 2.6274070739746094,
357
+ "step": 450
358
  },
359
  {
360
+ "epoch": 0.06571428571428571,
361
+ "grad_norm": 0.10693359375,
362
+ "learning_rate": 0.00029744376583970897,
363
+ "loss": 2.629538154602051,
364
+ "step": 460
365
  },
366
  {
367
+ "epoch": 0.06714285714285714,
368
+ "grad_norm": 0.111328125,
369
+ "learning_rate": 0.0002973176169353022,
370
+ "loss": 2.6362348556518556,
371
+ "step": 470
372
  },
373
  {
374
+ "epoch": 0.06857142857142857,
375
+ "grad_norm": 1.1015625,
376
+ "learning_rate": 0.00029718845790900785,
377
+ "loss": 2.613158416748047,
378
+ "step": 480
379
  },
380
  {
381
+ "epoch": 0.07,
382
+ "grad_norm": 0.412109375,
383
+ "learning_rate": 0.00029705629139991567,
384
+ "loss": 2.658156967163086,
385
+ "step": 490
386
  },
387
  {
388
+ "epoch": 0.07142857142857142,
389
+ "grad_norm": 0.42578125,
390
+ "learning_rate": 0.000296921120108567,
391
+ "loss": 2.6301057815551756,
392
+ "step": 500
393
  },
394
  {
395
+ "epoch": 0.07142857142857142,
396
+ "eval_loss": 3.0305709838867188,
397
+ "eval_runtime": 4.4742,
398
+ "eval_samples_per_second": 66.604,
399
+ "eval_steps_per_second": 3.8,
400
+ "step": 500
401
  }
402
  ],
403
+ "logging_steps": 10,
404
+ "max_steps": 7000,
405
  "num_input_tokens_seen": 0,
406
  "num_train_epochs": 9223372036854775807,
407
+ "save_steps": 500,
408
  "stateful_callbacks": {
409
  "TrainerControl": {
410
  "args": {
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4d0fa3459db214f81ac56235c0274e72db303ff14a77456f1fa013b5e4128f2e
3
- size 5265
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6f167c9a1f8ccc5220a546dd5b0acec4ebf0eadee91fee1ea2fa171733ead18e
3
+ size 5201