CodeIsAbstract commited on
Commit
463db0e
·
verified ·
1 Parent(s): e115d34

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f421426d14674c7d4ad546e19a4e57083d7bca462ab2e13d58cc41736dd8c7fe
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d91f361723cdb3da7122704a5a753827705ee51e8d9db8d95d17fa16e1214a1
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6ccea549a034e18c4f5ecfe1285a4528b857a0949f057eddfb3c088844cb9c36
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be60ae141d864e2c961c2d98241d3c95b42ab05195017b53e1dd1fb3455bc2c7
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ca0548a1b90eaf6590b9f6afac3530aaed1157198c5a087c4229ad15001933d9
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:76cd281a3ce995fc434ac757f1094066d955dd6b7188cf148f0dd1523cb9cd80
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2d251671b590d29a4724daa893ce2020a6083e1de3b017ea6154c342d8e856b9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dfc5c6949cf8a9eca4d88710f0c9c06f410fb63454fb60c644727889ad457648
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,331 +2,409 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.008,
6
- "eval_steps": 100,
7
- "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0002,
14
- "grad_norm": 23680.0,
15
- "learning_rate": 6e-05,
16
- "loss": 13.378594970703125,
17
- "step": 10
18
  },
19
  {
20
- "epoch": 0.0004,
21
- "grad_norm": 308.0,
22
- "learning_rate": 0.0001266666666666667,
23
- "loss": 13.249179077148437,
24
- "step": 20
25
  },
26
  {
27
- "epoch": 0.0006,
28
- "grad_norm": 1240.0,
29
- "learning_rate": 0.00019333333333333333,
30
- "loss": 13.185873413085938,
31
- "step": 30
32
  },
33
  {
34
- "epoch": 0.0008,
35
- "grad_norm": 54528.0,
36
- "learning_rate": 0.00026000000000000003,
37
- "loss": 12.9256103515625,
38
- "step": 40
39
  },
40
  {
41
- "epoch": 0.001,
42
- "grad_norm": 1120.0,
43
- "learning_rate": 0.0003266666666666667,
44
- "loss": 12.976683044433594,
45
- "step": 50
46
  },
47
  {
48
- "epoch": 0.0012,
49
- "grad_norm": 1012.0,
50
- "learning_rate": 0.0003933333333333333,
51
- "loss": 13.311146545410157,
52
- "step": 60
53
  },
54
  {
55
- "epoch": 0.0014,
56
- "grad_norm": 1192.0,
57
- "learning_rate": 0.00046,
58
- "loss": 13.249429321289062,
59
- "step": 70
60
  },
61
  {
62
- "epoch": 0.0016,
63
- "grad_norm": 560.0,
64
- "learning_rate": 0.0005266666666666666,
65
- "loss": 13.768452453613282,
66
- "step": 80
67
  },
68
  {
69
- "epoch": 0.0018,
70
- "grad_norm": 764.0,
71
- "learning_rate": 0.0005933333333333334,
72
- "loss": 14.640971374511718,
73
- "step": 90
74
  },
75
  {
76
- "epoch": 0.002,
77
- "grad_norm": 2736.0,
78
- "learning_rate": 0.00066,
79
- "loss": 14.4900634765625,
80
- "step": 100
81
  },
82
  {
83
- "epoch": 0.002,
84
- "eval_loss": 3.883816957473755,
85
- "eval_runtime": 4.54,
86
- "eval_samples_per_second": 65.638,
87
- "eval_steps_per_second": 3.744,
88
- "step": 100
89
  },
90
  {
91
- "epoch": 0.0022,
92
- "grad_norm": 5280.0,
93
- "learning_rate": 0.0007266666666666667,
94
- "loss": 14.381755065917968,
95
- "step": 110
96
  },
97
  {
98
- "epoch": 0.0024,
99
- "grad_norm": 298.0,
100
- "learning_rate": 0.0007933333333333334,
101
- "loss": 14.265592956542969,
102
- "step": 120
103
  },
104
  {
105
- "epoch": 0.0026,
106
- "grad_norm": 45.0,
107
- "learning_rate": 0.00086,
108
- "loss": 13.5162353515625,
109
- "step": 130
110
  },
111
  {
112
- "epoch": 0.0028,
113
- "grad_norm": 74.5,
114
- "learning_rate": 0.0009266666666666667,
115
- "loss": 13.169541931152343,
116
- "step": 140
117
  },
118
  {
119
- "epoch": 0.003,
120
- "grad_norm": 8.125,
121
- "learning_rate": 0.0009933333333333333,
122
- "loss": 12.071795654296874,
123
- "step": 150
124
  },
125
  {
126
- "epoch": 0.0032,
127
- "grad_norm": 2.703125,
128
- "learning_rate": 0.001,
129
- "loss": 11.409983825683593,
130
- "step": 160
131
  },
132
  {
133
- "epoch": 0.0034,
134
- "grad_norm": 1.03125,
135
- "learning_rate": 0.001,
136
- "loss": 10.876200103759766,
137
- "step": 170
138
  },
139
  {
140
- "epoch": 0.0036,
141
- "grad_norm": 0.91796875,
142
- "learning_rate": 0.001,
143
- "loss": 10.752876281738281,
144
- "step": 180
145
  },
146
  {
147
- "epoch": 0.0038,
148
- "grad_norm": 0.71484375,
149
- "learning_rate": 0.001,
150
- "loss": 10.67746810913086,
151
- "step": 190
152
  },
153
  {
154
- "epoch": 0.004,
155
- "grad_norm": 5.0,
156
- "learning_rate": 0.001,
157
- "loss": 10.72914810180664,
158
- "step": 200
159
  },
160
  {
161
- "epoch": 0.004,
162
- "eval_loss": 3.0593068599700928,
163
- "eval_runtime": 4.4647,
164
- "eval_samples_per_second": 66.746,
165
- "eval_steps_per_second": 3.808,
166
- "step": 200
167
  },
168
  {
169
- "epoch": 0.0042,
170
- "grad_norm": 12.25,
171
- "learning_rate": 0.001,
172
- "loss": 10.68040771484375,
173
- "step": 210
174
  },
175
  {
176
- "epoch": 0.0044,
177
- "grad_norm": 0.6640625,
178
- "learning_rate": 0.001,
179
- "loss": 10.56131820678711,
180
- "step": 220
181
  },
182
  {
183
- "epoch": 0.0046,
184
- "grad_norm": 1.0703125,
185
- "learning_rate": 0.001,
186
- "loss": 10.673944854736328,
187
- "step": 230
188
  },
189
  {
190
- "epoch": 0.0048,
191
- "grad_norm": 0.59375,
192
- "learning_rate": 0.001,
193
- "loss": 10.737371826171875,
194
- "step": 240
195
  },
196
  {
197
- "epoch": 0.005,
198
- "grad_norm": 0.546875,
199
- "learning_rate": 0.001,
200
- "loss": 10.560675048828125,
201
- "step": 250
202
  },
203
  {
204
- "epoch": 0.0052,
205
- "grad_norm": 0.470703125,
206
- "learning_rate": 0.001,
207
- "loss": 10.704891204833984,
208
- "step": 260
209
  },
210
  {
211
- "epoch": 0.0054,
212
- "grad_norm": 0.494140625,
213
- "learning_rate": 0.001,
214
- "loss": 10.544506072998047,
215
- "step": 270
216
  },
217
  {
218
- "epoch": 0.0056,
219
- "grad_norm": 0.421875,
220
- "learning_rate": 0.001,
221
- "loss": 10.619854736328126,
222
- "step": 280
223
  },
224
  {
225
- "epoch": 0.0058,
226
- "grad_norm": 44.5,
227
- "learning_rate": 0.001,
228
- "loss": 10.606621551513673,
229
- "step": 290
230
  },
231
  {
232
- "epoch": 0.006,
233
- "grad_norm": 0.400390625,
234
- "learning_rate": 0.001,
235
- "loss": 10.764106750488281,
236
- "step": 300
237
  },
238
  {
239
- "epoch": 0.006,
240
- "eval_loss": 3.049805164337158,
241
- "eval_runtime": 4.4123,
242
- "eval_samples_per_second": 67.538,
243
- "eval_steps_per_second": 3.853,
244
- "step": 300
245
  },
246
  {
247
- "epoch": 0.0062,
248
- "grad_norm": 0.45703125,
249
- "learning_rate": 0.001,
250
- "loss": 10.587713623046875,
251
- "step": 310
252
  },
253
  {
254
- "epoch": 0.0064,
255
- "grad_norm": 0.50390625,
256
- "learning_rate": 0.001,
257
- "loss": 10.587606811523438,
258
- "step": 320
259
  },
260
  {
261
- "epoch": 0.0066,
262
- "grad_norm": 0.373046875,
263
- "learning_rate": 0.001,
264
- "loss": 10.690779876708984,
265
- "step": 330
266
  },
267
  {
268
- "epoch": 0.0068,
269
- "grad_norm": 0.33984375,
270
- "learning_rate": 0.001,
271
- "loss": 10.637954711914062,
272
- "step": 340
273
  },
274
  {
275
- "epoch": 0.007,
276
- "grad_norm": 0.443359375,
277
- "learning_rate": 0.001,
278
- "loss": 10.537116241455077,
279
- "step": 350
280
  },
281
  {
282
- "epoch": 0.0072,
283
- "grad_norm": 0.5,
284
- "learning_rate": 0.001,
285
- "loss": 10.539633178710938,
286
- "step": 360
287
  },
288
  {
289
- "epoch": 0.0074,
290
- "grad_norm": 0.72265625,
291
- "learning_rate": 0.001,
292
- "loss": 10.701078033447265,
293
- "step": 370
294
  },
295
  {
296
- "epoch": 0.0076,
297
- "grad_norm": 0.36328125,
298
- "learning_rate": 0.001,
299
- "loss": 10.648773193359375,
300
- "step": 380
301
  },
302
  {
303
- "epoch": 0.0078,
304
- "grad_norm": 0.380859375,
305
- "learning_rate": 0.001,
306
- "loss": 10.477677154541016,
307
- "step": 390
308
  },
309
  {
310
- "epoch": 0.008,
311
- "grad_norm": 7.6875,
312
- "learning_rate": 0.001,
313
- "loss": 10.57955093383789,
314
- "step": 400
315
  },
316
  {
317
- "epoch": 0.008,
318
- "eval_loss": 3.0504510402679443,
319
- "eval_runtime": 4.4431,
320
- "eval_samples_per_second": 67.07,
321
- "eval_steps_per_second": 3.826,
322
- "step": 400
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
323
  }
324
  ],
325
- "logging_steps": 10,
326
- "max_steps": 50000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 9223372036854775807,
329
- "save_steps": 400,
330
  "stateful_callbacks": {
331
  "TrainerControl": {
332
  "args": {
@@ -339,7 +417,7 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 6.52309029715968e+16,
343
  "train_batch_size": 18,
344
  "trial_name": null,
345
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.05555555555555555,
6
+ "eval_steps": 200,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0011111111111111111,
14
+ "grad_norm": 95.5,
15
+ "learning_rate": 0.00011399999999999999,
16
+ "loss": 3.3263259887695313,
17
+ "step": 20
18
  },
19
  {
20
+ "epoch": 0.0022222222222222222,
21
+ "grad_norm": 37.75,
22
+ "learning_rate": 0.000234,
23
+ "loss": 3.246612548828125,
24
+ "step": 40
25
  },
26
  {
27
+ "epoch": 0.0033333333333333335,
28
+ "grad_norm": 72.0,
29
+ "learning_rate": 0.0003,
30
+ "loss": 3.197848892211914,
31
+ "step": 60
32
  },
33
  {
34
+ "epoch": 0.0044444444444444444,
35
+ "grad_norm": 137.0,
36
+ "learning_rate": 0.0003,
37
+ "loss": 3.0936569213867187,
38
+ "step": 80
39
  },
40
  {
41
+ "epoch": 0.005555555555555556,
42
+ "grad_norm": 13.8125,
43
+ "learning_rate": 0.0003,
44
+ "loss": 3.0532867431640627,
45
+ "step": 100
46
  },
47
  {
48
+ "epoch": 0.006666666666666667,
49
+ "grad_norm": 11.875,
50
+ "learning_rate": 0.0003,
51
+ "loss": 3.0499711990356446,
52
+ "step": 120
53
  },
54
  {
55
+ "epoch": 0.0077777777777777776,
56
+ "grad_norm": 34.0,
57
+ "learning_rate": 0.0003,
58
+ "loss": 2.9780717849731446,
59
+ "step": 140
60
  },
61
  {
62
+ "epoch": 0.008888888888888889,
63
+ "grad_norm": 16.375,
64
+ "learning_rate": 0.0003,
65
+ "loss": 2.9729740142822267,
66
+ "step": 160
67
  },
68
  {
69
+ "epoch": 0.01,
70
+ "grad_norm": 1.6640625,
71
+ "learning_rate": 0.0003,
72
+ "loss": 2.905041313171387,
73
+ "step": 180
74
  },
75
  {
76
+ "epoch": 0.011111111111111112,
77
+ "grad_norm": 2.5625,
78
+ "learning_rate": 0.0003,
79
+ "loss": 2.8482807159423826,
80
+ "step": 200
81
  },
82
  {
83
+ "epoch": 0.011111111111111112,
84
+ "eval_loss": 3.1418681144714355,
85
+ "eval_runtime": 4.509,
86
+ "eval_samples_per_second": 66.091,
87
+ "eval_steps_per_second": 3.77,
88
+ "step": 200
89
  },
90
  {
91
+ "epoch": 0.012222222222222223,
92
+ "grad_norm": 1.6875,
93
+ "learning_rate": 0.0003,
94
+ "loss": 2.763028144836426,
95
+ "step": 220
96
  },
97
  {
98
+ "epoch": 0.013333333333333334,
99
+ "grad_norm": 0.5078125,
100
+ "learning_rate": 0.0003,
101
+ "loss": 2.7428956985473634,
102
+ "step": 240
103
  },
104
  {
105
+ "epoch": 0.014444444444444444,
106
+ "grad_norm": 0.2578125,
107
+ "learning_rate": 0.0003,
108
+ "loss": 2.6967418670654295,
109
+ "step": 260
110
  },
111
  {
112
+ "epoch": 0.015555555555555555,
113
+ "grad_norm": 0.357421875,
114
+ "learning_rate": 0.0003,
115
+ "loss": 2.666821098327637,
116
+ "step": 280
117
  },
118
  {
119
+ "epoch": 0.016666666666666666,
120
+ "grad_norm": 0.404296875,
121
+ "learning_rate": 0.0003,
122
+ "loss": 2.6833669662475588,
123
+ "step": 300
124
  },
125
  {
126
+ "epoch": 0.017777777777777778,
127
+ "grad_norm": 0.142578125,
128
+ "learning_rate": 0.0003,
129
+ "loss": 2.6547067642211912,
130
+ "step": 320
131
  },
132
  {
133
+ "epoch": 0.01888888888888889,
134
+ "grad_norm": 0.10986328125,
135
+ "learning_rate": 0.0003,
136
+ "loss": 2.6726154327392577,
137
+ "step": 340
138
  },
139
  {
140
+ "epoch": 0.02,
141
+ "grad_norm": 0.310546875,
142
+ "learning_rate": 0.0003,
143
+ "loss": 2.639314079284668,
144
+ "step": 360
145
  },
146
  {
147
+ "epoch": 0.021111111111111112,
148
+ "grad_norm": 0.265625,
149
+ "learning_rate": 0.0003,
150
+ "loss": 2.6715606689453124,
151
+ "step": 380
152
  },
153
  {
154
+ "epoch": 0.022222222222222223,
155
+ "grad_norm": 0.15234375,
156
+ "learning_rate": 0.0003,
157
+ "loss": 2.633011817932129,
158
+ "step": 400
159
  },
160
  {
161
+ "epoch": 0.022222222222222223,
162
+ "eval_loss": 3.0355894565582275,
163
+ "eval_runtime": 4.4466,
164
+ "eval_samples_per_second": 67.017,
165
+ "eval_steps_per_second": 3.823,
166
+ "step": 400
167
  },
168
  {
169
+ "epoch": 0.023333333333333334,
170
+ "grad_norm": 0.9296875,
171
+ "learning_rate": 0.0003,
172
+ "loss": 2.6349510192871093,
173
+ "step": 420
174
  },
175
  {
176
+ "epoch": 0.024444444444444446,
177
+ "grad_norm": 0.25,
178
+ "learning_rate": 0.0003,
179
+ "loss": 2.6378190994262694,
180
+ "step": 440
181
  },
182
  {
183
+ "epoch": 0.025555555555555557,
184
+ "grad_norm": 0.091796875,
185
+ "learning_rate": 0.0003,
186
+ "loss": 2.645473670959473,
187
+ "step": 460
188
  },
189
  {
190
+ "epoch": 0.02666666666666667,
191
+ "grad_norm": 0.7109375,
192
+ "learning_rate": 0.0003,
193
+ "loss": 2.6475948333740233,
194
+ "step": 480
195
  },
196
  {
197
+ "epoch": 0.027777777777777776,
198
+ "grad_norm": 0.0927734375,
199
+ "learning_rate": 0.0003,
200
+ "loss": 2.652556228637695,
201
+ "step": 500
202
  },
203
  {
204
+ "epoch": 0.028888888888888888,
205
+ "grad_norm": 0.10400390625,
206
+ "learning_rate": 0.0003,
207
+ "loss": 2.6353578567504883,
208
+ "step": 520
209
  },
210
  {
211
+ "epoch": 0.03,
212
+ "grad_norm": 0.11767578125,
213
+ "learning_rate": 0.0003,
214
+ "loss": 2.617463493347168,
215
+ "step": 540
216
  },
217
  {
218
+ "epoch": 0.03111111111111111,
219
+ "grad_norm": 0.09619140625,
220
+ "learning_rate": 0.0003,
221
+ "loss": 2.6145917892456056,
222
+ "step": 560
223
  },
224
  {
225
+ "epoch": 0.03222222222222222,
226
+ "grad_norm": 0.291015625,
227
+ "learning_rate": 0.0003,
228
+ "loss": 2.6116378784179686,
229
+ "step": 580
230
  },
231
  {
232
+ "epoch": 0.03333333333333333,
233
+ "grad_norm": 0.09423828125,
234
+ "learning_rate": 0.0003,
235
+ "loss": 2.6324670791625975,
236
+ "step": 600
237
  },
238
  {
239
+ "epoch": 0.03333333333333333,
240
+ "eval_loss": 3.0325393676757812,
241
+ "eval_runtime": 4.4279,
242
+ "eval_samples_per_second": 67.301,
243
+ "eval_steps_per_second": 3.839,
244
+ "step": 600
245
  },
246
  {
247
+ "epoch": 0.034444444444444444,
248
+ "grad_norm": 0.09912109375,
249
+ "learning_rate": 0.0003,
250
+ "loss": 2.627296257019043,
251
+ "step": 620
252
  },
253
  {
254
+ "epoch": 0.035555555555555556,
255
+ "grad_norm": 0.1064453125,
256
+ "learning_rate": 0.0003,
257
+ "loss": 2.615768623352051,
258
+ "step": 640
259
  },
260
  {
261
+ "epoch": 0.03666666666666667,
262
+ "grad_norm": 0.0830078125,
263
+ "learning_rate": 0.0003,
264
+ "loss": 2.630397415161133,
265
+ "step": 660
266
  },
267
  {
268
+ "epoch": 0.03777777777777778,
269
+ "grad_norm": 2.34375,
270
+ "learning_rate": 0.0003,
271
+ "loss": 2.6295297622680662,
272
+ "step": 680
273
  },
274
  {
275
+ "epoch": 0.03888888888888889,
276
+ "grad_norm": 0.091796875,
277
+ "learning_rate": 0.0003,
278
+ "loss": 2.6139799118041993,
279
+ "step": 700
280
  },
281
  {
282
+ "epoch": 0.04,
283
+ "grad_norm": 0.177734375,
284
+ "learning_rate": 0.0003,
285
+ "loss": 2.612909507751465,
286
+ "step": 720
287
  },
288
  {
289
+ "epoch": 0.04111111111111111,
290
+ "grad_norm": 0.08984375,
291
+ "learning_rate": 0.0003,
292
+ "loss": 2.637497901916504,
293
+ "step": 740
294
  },
295
  {
296
+ "epoch": 0.042222222222222223,
297
+ "grad_norm": 0.091796875,
298
+ "learning_rate": 0.0003,
299
+ "loss": 2.6311773300170898,
300
+ "step": 760
301
  },
302
  {
303
+ "epoch": 0.043333333333333335,
304
+ "grad_norm": 0.1748046875,
305
+ "learning_rate": 0.0003,
306
+ "loss": 2.631211280822754,
307
+ "step": 780
308
  },
309
  {
310
+ "epoch": 0.044444444444444446,
311
+ "grad_norm": 0.08154296875,
312
+ "learning_rate": 0.0003,
313
+ "loss": 2.64044075012207,
314
+ "step": 800
315
  },
316
  {
317
+ "epoch": 0.044444444444444446,
318
+ "eval_loss": 3.029780149459839,
319
+ "eval_runtime": 4.4461,
320
+ "eval_samples_per_second": 67.025,
321
+ "eval_steps_per_second": 3.824,
322
+ "step": 800
323
+ },
324
+ {
325
+ "epoch": 0.04555555555555556,
326
+ "grad_norm": 0.0810546875,
327
+ "learning_rate": 0.0003,
328
+ "loss": 2.6481130599975584,
329
+ "step": 820
330
+ },
331
+ {
332
+ "epoch": 0.04666666666666667,
333
+ "grad_norm": 0.06884765625,
334
+ "learning_rate": 0.0003,
335
+ "loss": 2.6261564254760743,
336
+ "step": 840
337
+ },
338
+ {
339
+ "epoch": 0.04777777777777778,
340
+ "grad_norm": 0.07373046875,
341
+ "learning_rate": 0.0003,
342
+ "loss": 2.611224365234375,
343
+ "step": 860
344
+ },
345
+ {
346
+ "epoch": 0.04888888888888889,
347
+ "grad_norm": 0.119140625,
348
+ "learning_rate": 0.0003,
349
+ "loss": 2.636740875244141,
350
+ "step": 880
351
+ },
352
+ {
353
+ "epoch": 0.05,
354
+ "grad_norm": 0.06884765625,
355
+ "learning_rate": 0.0003,
356
+ "loss": 2.614007568359375,
357
+ "step": 900
358
+ },
359
+ {
360
+ "epoch": 0.051111111111111114,
361
+ "grad_norm": 0.068359375,
362
+ "learning_rate": 0.0003,
363
+ "loss": 2.616529846191406,
364
+ "step": 920
365
+ },
366
+ {
367
+ "epoch": 0.052222222222222225,
368
+ "grad_norm": 0.07470703125,
369
+ "learning_rate": 0.0003,
370
+ "loss": 2.6233530044555664,
371
+ "step": 940
372
+ },
373
+ {
374
+ "epoch": 0.05333333333333334,
375
+ "grad_norm": 0.06689453125,
376
+ "learning_rate": 0.0003,
377
+ "loss": 2.600985527038574,
378
+ "step": 960
379
+ },
380
+ {
381
+ "epoch": 0.05444444444444444,
382
+ "grad_norm": 0.0791015625,
383
+ "learning_rate": 0.0003,
384
+ "loss": 2.6462072372436523,
385
+ "step": 980
386
+ },
387
+ {
388
+ "epoch": 0.05555555555555555,
389
+ "grad_norm": 0.07470703125,
390
+ "learning_rate": 0.0003,
391
+ "loss": 2.6183977127075195,
392
+ "step": 1000
393
+ },
394
+ {
395
+ "epoch": 0.05555555555555555,
396
+ "eval_loss": 3.0301759243011475,
397
+ "eval_runtime": 4.4446,
398
+ "eval_samples_per_second": 67.048,
399
+ "eval_steps_per_second": 3.825,
400
+ "step": 1000
401
  }
402
  ],
403
+ "logging_steps": 20,
404
+ "max_steps": 18000,
405
  "num_input_tokens_seen": 0,
406
  "num_train_epochs": 9223372036854775807,
407
+ "save_steps": 1000,
408
  "stateful_callbacks": {
409
  "TrainerControl": {
410
  "args": {
 
417
  "attributes": {}
418
  }
419
  },
420
+ "total_flos": 1.63077257428992e+17,
421
  "train_batch_size": 18,
422
  "trial_name": null,
423
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:666419280b80b3d3942f48be91e30de30a6906e1b7b476ee09f4d3f438f4356a
3
  size 5265
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d0fa3459db214f81ac56235c0274e72db303ff14a77456f1fa013b5e4128f2e
3
  size 5265