CodeIsAbstract commited on
Commit
2cc37bf
·
verified ·
1 Parent(s): e8bce90

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:95c73d234eccc64a09618013c53a92bc4a7c429ebfb1e2678f1bc84f054f04ca
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:561331e81c6fcc3ea4bd80e90aeeb451e6195fe512d776202f16932e0c17ea6e
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dde59450052d104d8238c08fc21017ca7929ea8c3847f36800fa1a9fca90e493
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7e830fa46b8c9ec45ddf2b32b75587588f6e3d46a65f46e08aad400084a9a6d5
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1f19cd3c4a3ac433efb0482e01f74c9bb46c2bbfbf5afed45038fc450ceaa5f6
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9842a5fdf27ba02b590d63e6f1d65edf706e29451025042d306b57ac70d681d7
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0b3af4aaa142e4a55ebe50a8d2db120dc2b7b23b5e6bb84de2f69f7d1627803e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:62e2260209115d04c8ea31c913d8d38246600d352056e2ec48e197265ec64063
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,331 +2,175 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.03636363636363636,
6
- "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0009090909090909091,
14
- "grad_norm": 0.25,
15
- "learning_rate": 0.00066,
16
- "loss": 2.65932861328125,
17
- "step": 100
18
- },
19
- {
20
- "epoch": 0.0018181818181818182,
21
- "grad_norm": 0.193359375,
22
- "learning_rate": 0.000999999509056267,
23
- "loss": 2.663287353515625,
24
- "step": 200
25
- },
26
- {
27
- "epoch": 0.0027272727272727275,
28
- "grad_norm": 0.19140625,
29
- "learning_rate": 0.0009999954604635122,
30
- "loss": 2.6766876220703124,
31
- "step": 300
32
- },
33
- {
34
- "epoch": 0.0036363636363636364,
35
- "grad_norm": 0.1630859375,
36
- "learning_rate": 0.0009999873224161853,
37
- "loss": 2.6875625610351563,
38
- "step": 400
39
- },
40
- {
41
- "epoch": 0.004545454545454545,
42
- "grad_norm": 0.1591796875,
43
- "learning_rate": 0.000999975094980847,
44
- "loss": 2.6895318603515626,
45
- "step": 500
46
- },
47
- {
48
- "epoch": 0.005454545454545455,
49
- "grad_norm": 0.16796875,
50
- "learning_rate": 0.0009999587782575051,
51
- "loss": 2.7013665771484376,
52
- "step": 600
53
- },
54
- {
55
- "epoch": 0.006363636363636364,
56
- "grad_norm": 0.162109375,
57
- "learning_rate": 0.0009999383723796143,
58
- "loss": 2.6975772094726564,
59
- "step": 700
60
- },
61
- {
62
- "epoch": 0.007272727272727273,
63
- "grad_norm": 0.17578125,
64
- "learning_rate": 0.0009999138775140734,
65
- "loss": 2.694542236328125,
66
- "step": 800
67
- },
68
- {
69
- "epoch": 0.008181818181818182,
70
- "grad_norm": 0.181640625,
71
- "learning_rate": 0.000999885293861226,
72
- "loss": 2.6991891479492187,
73
- "step": 900
74
- },
75
- {
76
- "epoch": 0.00909090909090909,
77
- "grad_norm": 0.1787109375,
78
- "learning_rate": 0.0009998526216548572,
79
- "loss": 2.6621365356445312,
80
- "step": 1000
81
- },
82
- {
83
- "epoch": 0.00909090909090909,
84
- "eval_loss": 3.092919111251831,
85
- "eval_runtime": 5.3241,
86
- "eval_samples_per_second": 112.132,
87
- "eval_steps_per_second": 28.174,
88
- "step": 1000
89
- },
90
- {
91
- "epoch": 0.01,
92
- "grad_norm": 0.1591796875,
93
- "learning_rate": 0.0009998158611621922,
94
- "loss": 2.6619110107421875,
95
- "step": 1100
96
- },
97
- {
98
- "epoch": 0.01090909090909091,
99
- "grad_norm": 0.1669921875,
100
- "learning_rate": 0.0009997750126838946,
101
- "loss": 2.700511779785156,
102
- "step": 1200
103
- },
104
- {
105
- "epoch": 0.011818181818181818,
106
- "grad_norm": 0.1591796875,
107
- "learning_rate": 0.0009997300765540635,
108
- "loss": 2.689774475097656,
109
- "step": 1300
110
- },
111
- {
112
- "epoch": 0.012727272727272728,
113
- "grad_norm": 0.171875,
114
- "learning_rate": 0.0009996810531402306,
115
- "loss": 2.69824462890625,
116
- "step": 1400
117
- },
118
- {
119
- "epoch": 0.013636363636363636,
120
- "grad_norm": 0.166015625,
121
- "learning_rate": 0.0009996279428433578,
122
- "loss": 2.7036367797851564,
123
- "step": 1500
124
- },
125
- {
126
- "epoch": 0.014545454545454545,
127
- "grad_norm": 0.1591796875,
128
- "learning_rate": 0.000999570746097833,
129
- "loss": 2.6719509887695314,
130
- "step": 1600
131
- },
132
- {
133
- "epoch": 0.015454545454545455,
134
- "grad_norm": 0.16796875,
135
- "learning_rate": 0.0009995094633714677,
136
- "loss": 2.6892950439453127,
137
- "step": 1700
138
- },
139
- {
140
- "epoch": 0.016363636363636365,
141
- "grad_norm": 0.23828125,
142
- "learning_rate": 0.0009994440951654924,
143
- "loss": 2.7064022827148437,
144
- "step": 1800
145
  },
146
  {
147
- "epoch": 0.017272727272727273,
148
- "grad_norm": 0.169921875,
149
- "learning_rate": 0.0009993746420145523,
150
- "loss": 2.6744000244140627,
151
- "step": 1900
152
  },
153
  {
154
- "epoch": 0.01818181818181818,
155
- "grad_norm": 0.1630859375,
156
- "learning_rate": 0.0009993011044867038,
157
- "loss": 2.69939453125,
158
- "step": 2000
159
  },
160
  {
161
- "epoch": 0.01818181818181818,
162
- "eval_loss": 3.0854060649871826,
163
- "eval_runtime": 5.237,
164
- "eval_samples_per_second": 113.996,
165
- "eval_steps_per_second": 28.642,
166
- "step": 2000
167
  },
168
  {
169
- "epoch": 0.019090909090909092,
170
- "grad_norm": 0.150390625,
171
- "learning_rate": 0.000999223483183409,
172
- "loss": 2.6586599731445313,
173
- "step": 2100
174
  },
175
  {
176
- "epoch": 0.02,
177
- "grad_norm": 0.1572265625,
178
- "learning_rate": 0.000999141778739531,
179
- "loss": 2.689534912109375,
180
- "step": 2200
181
  },
182
  {
183
- "epoch": 0.02090909090909091,
184
- "grad_norm": 0.177734375,
185
- "learning_rate": 0.0009990559918233294,
186
- "loss": 2.6719708251953125,
187
- "step": 2300
188
  },
189
  {
190
- "epoch": 0.02181818181818182,
191
- "grad_norm": 0.20703125,
192
- "learning_rate": 0.0009989661231364542,
193
- "loss": 2.6797967529296876,
194
- "step": 2400
195
  },
196
  {
197
- "epoch": 0.022727272727272728,
198
- "grad_norm": 0.173828125,
199
- "learning_rate": 0.0009988721734139395,
200
- "loss": 2.68374267578125,
201
- "step": 2500
202
  },
203
  {
204
- "epoch": 0.023636363636363636,
205
- "grad_norm": 0.1748046875,
206
- "learning_rate": 0.0009987741434241983,
207
- "loss": 2.677032775878906,
208
- "step": 2600
209
- },
210
- {
211
- "epoch": 0.024545454545454544,
212
- "grad_norm": 0.2119140625,
213
- "learning_rate": 0.000998672033969017,
214
- "loss": 2.7009921264648438,
215
- "step": 2700
216
- },
217
- {
218
- "epoch": 0.025454545454545455,
219
- "grad_norm": 0.189453125,
220
- "learning_rate": 0.0009985658458835465,
221
- "loss": 2.66544677734375,
222
- "step": 2800
223
- },
224
- {
225
- "epoch": 0.026363636363636363,
226
- "grad_norm": 0.16796875,
227
- "learning_rate": 0.0009984555800362977,
228
- "loss": 2.6626568603515626,
229
- "step": 2900
230
- },
231
- {
232
- "epoch": 0.02727272727272727,
233
- "grad_norm": 0.1689453125,
234
- "learning_rate": 0.0009983412373291332,
235
- "loss": 2.708771057128906,
236
- "step": 3000
237
  },
238
  {
239
- "epoch": 0.02727272727272727,
240
- "eval_loss": 3.0887579917907715,
241
- "eval_runtime": 5.2307,
242
- "eval_samples_per_second": 114.134,
243
- "eval_steps_per_second": 28.677,
244
- "step": 3000
245
  },
246
  {
247
- "epoch": 0.028181818181818183,
248
- "grad_norm": 0.169921875,
249
- "learning_rate": 0.0009982228186972597,
250
- "loss": 2.6861444091796876,
251
- "step": 3100
252
  },
253
  {
254
- "epoch": 0.02909090909090909,
255
- "grad_norm": 0.19921875,
256
- "learning_rate": 0.0009981003251092215,
257
- "loss": 2.706370544433594,
258
- "step": 3200
259
  },
260
  {
261
- "epoch": 0.03,
262
- "grad_norm": 0.1708984375,
263
- "learning_rate": 0.0009979737575668917,
264
- "loss": 2.6617535400390624,
265
- "step": 3300
266
  },
267
  {
268
- "epoch": 0.03090909090909091,
269
- "grad_norm": 0.162109375,
270
- "learning_rate": 0.000997843117105464,
271
- "loss": 2.673968505859375,
272
- "step": 3400
273
  },
274
  {
275
- "epoch": 0.031818181818181815,
276
- "grad_norm": 0.1533203125,
277
- "learning_rate": 0.0009977084047934448,
278
- "loss": 2.6613540649414062,
279
- "step": 3500
280
  },
281
  {
282
- "epoch": 0.03272727272727273,
283
- "grad_norm": 0.1640625,
284
- "learning_rate": 0.0009975696217326435,
285
- "loss": 2.6789846801757813,
286
- "step": 3600
287
  },
288
  {
289
- "epoch": 0.03363636363636364,
290
- "grad_norm": 0.1787109375,
291
- "learning_rate": 0.0009974267690581644,
292
- "loss": 2.664853210449219,
293
- "step": 3700
294
  },
295
  {
296
- "epoch": 0.034545454545454546,
297
- "grad_norm": 0.1787109375,
298
- "learning_rate": 0.0009972798479383977,
299
- "loss": 2.685954284667969,
300
- "step": 3800
301
  },
302
  {
303
- "epoch": 0.035454545454545454,
304
- "grad_norm": 0.205078125,
305
- "learning_rate": 0.0009971288595750083,
306
- "loss": 2.677633056640625,
307
- "step": 3900
308
  },
309
  {
310
- "epoch": 0.03636363636363636,
311
- "grad_norm": 0.1630859375,
312
- "learning_rate": 0.0009969738052029277,
313
- "loss": 2.6985595703125,
314
- "step": 4000
315
  },
316
  {
317
- "epoch": 0.03636363636363636,
318
- "eval_loss": 3.0944459438323975,
319
- "eval_runtime": 5.3684,
320
- "eval_samples_per_second": 111.206,
321
- "eval_steps_per_second": 27.941,
322
- "step": 4000
323
  }
324
  ],
325
- "logging_steps": 100,
326
- "max_steps": 110000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 9223372036854775807,
329
- "save_steps": 4000,
330
  "stateful_callbacks": {
331
  "TrainerControl": {
332
  "args": {
@@ -339,8 +183,8 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 9.9658323984384e+16,
343
- "train_batch_size": 22,
344
  "trial_name": null,
345
  "trial_params": null
346
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.004,
6
+ "eval_steps": 100,
7
+ "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0002,
14
+ "grad_norm": 2672.0,
15
+ "learning_rate": 6e-05,
16
+ "loss": 26.674310302734376,
17
+ "step": 10
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
18
  },
19
  {
20
+ "epoch": 0.0004,
21
+ "grad_norm": 652.0,
22
+ "learning_rate": 0.0001266666666666667,
23
+ "loss": 26.47657165527344,
24
+ "step": 20
25
  },
26
  {
27
+ "epoch": 0.0006,
28
+ "grad_norm": 8256.0,
29
+ "learning_rate": 0.00019333333333333333,
30
+ "loss": 26.448770141601564,
31
+ "step": 30
32
  },
33
  {
34
+ "epoch": 0.0008,
35
+ "grad_norm": 13632.0,
36
+ "learning_rate": 0.00026000000000000003,
37
+ "loss": 25.869580078125,
38
+ "step": 40
 
39
  },
40
  {
41
+ "epoch": 0.001,
42
+ "grad_norm": 576.0,
43
+ "learning_rate": 0.0003266666666666667,
44
+ "loss": 25.283935546875,
45
+ "step": 50
46
  },
47
  {
48
+ "epoch": 0.0012,
49
+ "grad_norm": 212.0,
50
+ "learning_rate": 0.0003933333333333333,
51
+ "loss": 25.18727264404297,
52
+ "step": 60
53
  },
54
  {
55
+ "epoch": 0.0014,
56
+ "grad_norm": 109.5,
57
+ "learning_rate": 0.00046,
58
+ "loss": 24.912762451171876,
59
+ "step": 70
60
  },
61
  {
62
+ "epoch": 0.0016,
63
+ "grad_norm": 195.0,
64
+ "learning_rate": 0.0005266666666666666,
65
+ "loss": 24.746240234375,
66
+ "step": 80
67
  },
68
  {
69
+ "epoch": 0.0018,
70
+ "grad_norm": 1032.0,
71
+ "learning_rate": 0.0005933333333333334,
72
+ "loss": 24.107341003417968,
73
+ "step": 90
74
  },
75
  {
76
+ "epoch": 0.002,
77
+ "grad_norm": 135.0,
78
+ "learning_rate": 0.00066,
79
+ "loss": 23.6096923828125,
80
+ "step": 100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
81
  },
82
  {
83
+ "epoch": 0.002,
84
+ "eval_loss": 3.206618070602417,
85
+ "eval_runtime": 4.5304,
86
+ "eval_samples_per_second": 65.777,
87
+ "eval_steps_per_second": 3.752,
88
+ "step": 100
89
  },
90
  {
91
+ "epoch": 0.0022,
92
+ "grad_norm": 55.75,
93
+ "learning_rate": 0.0007266666666666667,
94
+ "loss": 22.759126281738283,
95
+ "step": 110
96
  },
97
  {
98
+ "epoch": 0.0024,
99
+ "grad_norm": 8.3125,
100
+ "learning_rate": 0.0007933333333333334,
101
+ "loss": 22.43407287597656,
102
+ "step": 120
103
  },
104
  {
105
+ "epoch": 0.0026,
106
+ "grad_norm": 2.15625,
107
+ "learning_rate": 0.00086,
108
+ "loss": 21.722286987304688,
109
+ "step": 130
110
  },
111
  {
112
+ "epoch": 0.0028,
113
+ "grad_norm": 1.046875,
114
+ "learning_rate": 0.0009266666666666667,
115
+ "loss": 21.480882263183595,
116
+ "step": 140
117
  },
118
  {
119
+ "epoch": 0.003,
120
+ "grad_norm": 1.1875,
121
+ "learning_rate": 0.0009933333333333333,
122
+ "loss": 21.57379913330078,
123
+ "step": 150
124
  },
125
  {
126
+ "epoch": 0.0032,
127
+ "grad_norm": 0.86328125,
128
+ "learning_rate": 0.001,
129
+ "loss": 21.262675476074218,
130
+ "step": 160
131
  },
132
  {
133
+ "epoch": 0.0034,
134
+ "grad_norm": 0.703125,
135
+ "learning_rate": 0.001,
136
+ "loss": 21.385879516601562,
137
+ "step": 170
138
  },
139
  {
140
+ "epoch": 0.0036,
141
+ "grad_norm": 0.84765625,
142
+ "learning_rate": 0.001,
143
+ "loss": 21.12591552734375,
144
+ "step": 180
145
  },
146
  {
147
+ "epoch": 0.0038,
148
+ "grad_norm": 0.78125,
149
+ "learning_rate": 0.001,
150
+ "loss": 21.397132873535156,
151
+ "step": 190
152
  },
153
  {
154
+ "epoch": 0.004,
155
+ "grad_norm": 0.67578125,
156
+ "learning_rate": 0.001,
157
+ "loss": 21.077099609375,
158
+ "step": 200
159
  },
160
  {
161
+ "epoch": 0.004,
162
+ "eval_loss": 3.0551397800445557,
163
+ "eval_runtime": 4.4525,
164
+ "eval_samples_per_second": 66.929,
165
+ "eval_steps_per_second": 3.818,
166
+ "step": 200
167
  }
168
  ],
169
+ "logging_steps": 10,
170
+ "max_steps": 50000,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 9223372036854775807,
173
+ "save_steps": 200,
174
  "stateful_callbacks": {
175
  "TrainerControl": {
176
  "args": {
 
183
  "attributes": {}
184
  }
185
  },
186
+ "total_flos": 6.52309029715968e+16,
187
+ "train_batch_size": 18,
188
  "trial_name": null,
189
  "trial_params": null
190
  }
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0e603921b5fe0b842c30ab2079e7b2171b7aa484c50ff53b17a53468fcd71ca0
3
- size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:668a0e377ebad57c5d1965a2d2cc09bbdbb8e4a54e766f286eae108d1e754ff9
3
+ size 5265