CodeIsAbstract commited on
Commit
a2497c0
·
verified ·
1 Parent(s): 45631c3

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e5742327392312dd1a7e23ca028cd9e6187add4ba8fb99e3a28ce17d8dbe3b26
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7818cb9e8dda5e969179fa08e8dd35b115d5074a67004905fa566419d85bb77
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d2a1a873a10876a3541cbda1bda6ef5a52202b10618859918417990065cc8b5c
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8dc22d80526ad400ab0440e70c39e50b019ae3f83f519ccbb65d3d05ae38552d
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:78f00cbf2a3a4305cbfbe0a453688d1bb898224da36be2f067d77db0d73462c1
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca0548a1b90eaf6590b9f6afac3530aaed1157198c5a087c4229ad15001933d9
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4dcd0fcafdf0cac56c6781c9554680e5d2ba9bfee0d823975791d1d648a00172
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:17980325c36d869e063d721782688706e0e30902ea32568d55af1c62cc6c3e6f
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -11,314 +11,314 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.0016666666666666668,
14
- "grad_norm": 0.140625,
15
- "learning_rate": 1.6000000000000003e-05,
16
- "loss": 2.6374650955200196,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.0033333333333333335,
21
- "grad_norm": 0.050537109375,
22
- "learning_rate": 3.6e-05,
23
- "loss": 2.6133955001831053,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.005,
28
- "grad_norm": 0.03466796875,
29
- "learning_rate": 5.6000000000000006e-05,
30
- "loss": 2.6178274154663086,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.006666666666666667,
35
- "grad_norm": 0.037109375,
36
- "learning_rate": 7.6e-05,
37
- "loss": 2.624565315246582,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.008333333333333333,
42
- "grad_norm": 0.6328125,
43
- "learning_rate": 9.6e-05,
44
- "loss": 2.619654655456543,
45
  "step": 25
46
  },
47
  {
48
  "epoch": 0.01,
49
- "grad_norm": 0.036376953125,
50
- "learning_rate": 0.0001,
51
- "loss": 2.609416389465332,
52
  "step": 30
53
  },
54
  {
55
  "epoch": 0.011666666666666667,
56
- "grad_norm": 0.06591796875,
57
- "learning_rate": 0.0001,
58
- "loss": 2.6385807037353515,
59
  "step": 35
60
  },
61
  {
62
  "epoch": 0.013333333333333334,
63
- "grad_norm": 0.041748046875,
64
- "learning_rate": 0.0001,
65
- "loss": 2.6341808319091795,
66
  "step": 40
67
  },
68
  {
69
  "epoch": 0.015,
70
- "grad_norm": 0.03662109375,
71
- "learning_rate": 0.0001,
72
- "loss": 2.6163944244384765,
73
  "step": 45
74
  },
75
  {
76
  "epoch": 0.016666666666666666,
77
- "grad_norm": 0.50390625,
78
- "learning_rate": 0.0001,
79
- "loss": 2.620637893676758,
80
  "step": 50
81
  },
82
  {
83
  "epoch": 0.016666666666666666,
84
- "eval_loss": 3.0298774242401123,
85
- "eval_runtime": 4.5343,
86
- "eval_samples_per_second": 65.722,
87
- "eval_steps_per_second": 3.749,
88
  "step": 50
89
  },
90
  {
91
  "epoch": 0.018333333333333333,
92
- "grad_norm": 0.1689453125,
93
- "learning_rate": 0.0001,
94
- "loss": 2.6260648727416993,
95
  "step": 55
96
  },
97
  {
98
  "epoch": 0.02,
99
- "grad_norm": 0.047607421875,
100
- "learning_rate": 0.0001,
101
- "loss": 2.6460687637329103,
102
  "step": 60
103
  },
104
  {
105
  "epoch": 0.021666666666666667,
106
- "grad_norm": 0.0380859375,
107
- "learning_rate": 0.0001,
108
- "loss": 2.6157953262329103,
109
  "step": 65
110
  },
111
  {
112
  "epoch": 0.023333333333333334,
113
- "grad_norm": 0.0986328125,
114
- "learning_rate": 0.0001,
115
- "loss": 2.61843318939209,
116
  "step": 70
117
  },
118
  {
119
  "epoch": 0.025,
120
- "grad_norm": 0.04931640625,
121
- "learning_rate": 0.0001,
122
- "loss": 2.651448059082031,
123
  "step": 75
124
  },
125
  {
126
  "epoch": 0.02666666666666667,
127
- "grad_norm": 0.038330078125,
128
- "learning_rate": 0.0001,
129
- "loss": 2.59826545715332,
130
  "step": 80
131
  },
132
  {
133
  "epoch": 0.028333333333333332,
134
- "grad_norm": 0.042724609375,
135
- "learning_rate": 0.0001,
136
- "loss": 2.6349016189575196,
137
  "step": 85
138
  },
139
  {
140
  "epoch": 0.03,
141
- "grad_norm": 0.036376953125,
142
- "learning_rate": 0.0001,
143
- "loss": 2.6166595458984374,
144
  "step": 90
145
  },
146
  {
147
  "epoch": 0.03166666666666667,
148
- "grad_norm": 0.74609375,
149
- "learning_rate": 0.0001,
150
- "loss": 2.6394269943237303,
151
  "step": 95
152
  },
153
  {
154
  "epoch": 0.03333333333333333,
155
- "grad_norm": 0.032958984375,
156
- "learning_rate": 0.0001,
157
- "loss": 2.6080867767333986,
158
  "step": 100
159
  },
160
  {
161
  "epoch": 0.03333333333333333,
162
- "eval_loss": 3.0300662517547607,
163
- "eval_runtime": 4.4357,
164
- "eval_samples_per_second": 67.181,
165
- "eval_steps_per_second": 3.832,
166
  "step": 100
167
  },
168
  {
169
  "epoch": 0.035,
170
- "grad_norm": 0.036376953125,
171
- "learning_rate": 0.0001,
172
- "loss": 2.6302499771118164,
173
  "step": 105
174
  },
175
  {
176
  "epoch": 0.03666666666666667,
177
- "grad_norm": 0.042724609375,
178
- "learning_rate": 0.0001,
179
- "loss": 2.646461296081543,
180
  "step": 110
181
  },
182
  {
183
  "epoch": 0.03833333333333333,
184
- "grad_norm": 0.05517578125,
185
- "learning_rate": 0.0001,
186
- "loss": 2.6076818466186524,
187
  "step": 115
188
  },
189
  {
190
  "epoch": 0.04,
191
- "grad_norm": 0.032470703125,
192
- "learning_rate": 0.0001,
193
- "loss": 2.616958236694336,
194
  "step": 120
195
  },
196
  {
197
  "epoch": 0.041666666666666664,
198
- "grad_norm": 0.12353515625,
199
- "learning_rate": 0.0001,
200
- "loss": 2.603002166748047,
201
  "step": 125
202
  },
203
  {
204
  "epoch": 0.043333333333333335,
205
- "grad_norm": 0.034912109375,
206
- "learning_rate": 0.0001,
207
- "loss": 2.6103355407714846,
208
  "step": 130
209
  },
210
  {
211
  "epoch": 0.045,
212
- "grad_norm": 1.1484375,
213
- "learning_rate": 0.0001,
214
- "loss": 2.599276542663574,
215
  "step": 135
216
  },
217
  {
218
  "epoch": 0.04666666666666667,
219
- "grad_norm": 0.034912109375,
220
- "learning_rate": 0.0001,
221
- "loss": 2.590505599975586,
222
  "step": 140
223
  },
224
  {
225
  "epoch": 0.04833333333333333,
226
- "grad_norm": 0.03857421875,
227
- "learning_rate": 0.0001,
228
- "loss": 2.627636528015137,
229
  "step": 145
230
  },
231
  {
232
  "epoch": 0.05,
233
- "grad_norm": 262.0,
234
- "learning_rate": 0.0001,
235
- "loss": 2.6068201065063477,
236
  "step": 150
237
  },
238
  {
239
  "epoch": 0.05,
240
- "eval_loss": 3.0302553176879883,
241
- "eval_runtime": 4.3841,
242
- "eval_samples_per_second": 67.972,
243
- "eval_steps_per_second": 3.878,
244
  "step": 150
245
  },
246
  {
247
  "epoch": 0.051666666666666666,
248
- "grad_norm": 0.359375,
249
- "learning_rate": 0.0001,
250
- "loss": 2.617295265197754,
251
  "step": 155
252
  },
253
  {
254
  "epoch": 0.05333333333333334,
255
- "grad_norm": 0.078125,
256
- "learning_rate": 0.0001,
257
- "loss": 2.601869010925293,
258
  "step": 160
259
  },
260
  {
261
  "epoch": 0.055,
262
- "grad_norm": 0.06494140625,
263
- "learning_rate": 0.0001,
264
- "loss": 2.613315391540527,
265
  "step": 165
266
  },
267
  {
268
  "epoch": 0.056666666666666664,
269
- "grad_norm": 0.039794921875,
270
- "learning_rate": 0.0001,
271
- "loss": 2.601426124572754,
272
  "step": 170
273
  },
274
  {
275
  "epoch": 0.058333333333333334,
276
- "grad_norm": 0.035888671875,
277
- "learning_rate": 0.0001,
278
- "loss": 2.5963130950927735,
279
  "step": 175
280
  },
281
  {
282
  "epoch": 0.06,
283
- "grad_norm": 0.036376953125,
284
- "learning_rate": 0.0001,
285
- "loss": 2.614262008666992,
286
  "step": 180
287
  },
288
  {
289
  "epoch": 0.06166666666666667,
290
- "grad_norm": 0.0380859375,
291
- "learning_rate": 0.0001,
292
- "loss": 2.6094554901123046,
293
  "step": 185
294
  },
295
  {
296
  "epoch": 0.06333333333333334,
297
- "grad_norm": 0.064453125,
298
- "learning_rate": 0.0001,
299
- "loss": 2.6106592178344727,
300
  "step": 190
301
  },
302
  {
303
  "epoch": 0.065,
304
- "grad_norm": 0.037353515625,
305
- "learning_rate": 0.0001,
306
- "loss": 2.604322814941406,
307
  "step": 195
308
  },
309
  {
310
  "epoch": 0.06666666666666667,
311
- "grad_norm": 0.03955078125,
312
- "learning_rate": 0.0001,
313
- "loss": 2.5981882095336912,
314
  "step": 200
315
  },
316
  {
317
  "epoch": 0.06666666666666667,
318
- "eval_loss": 3.0292723178863525,
319
- "eval_runtime": 4.3967,
320
- "eval_samples_per_second": 67.777,
321
- "eval_steps_per_second": 3.866,
322
  "step": 200
323
  }
324
  ],
@@ -339,8 +339,8 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 1.304618059431936e+17,
343
- "train_batch_size": 18,
344
  "trial_name": null,
345
  "trial_params": null
346
  }
 
11
  "log_history": [
12
  {
13
  "epoch": 0.0016666666666666668,
14
+ "grad_norm": 0.419921875,
15
+ "learning_rate": 0.0002,
16
+ "loss": 2.7259506225585937,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.0033333333333333335,
21
+ "grad_norm": 0.326171875,
22
+ "learning_rate": 0.00045000000000000004,
23
+ "loss": 2.7605268478393556,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.005,
28
+ "grad_norm": 0.490234375,
29
+ "learning_rate": 0.0005,
30
+ "loss": 2.717021179199219,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.006666666666666667,
35
+ "grad_norm": 0.267578125,
36
+ "learning_rate": 0.0005,
37
+ "loss": 2.661177635192871,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.008333333333333333,
42
+ "grad_norm": 0.43359375,
43
+ "learning_rate": 0.0005,
44
+ "loss": 2.671949768066406,
45
  "step": 25
46
  },
47
  {
48
  "epoch": 0.01,
49
+ "grad_norm": 0.10595703125,
50
+ "learning_rate": 0.0005,
51
+ "loss": 2.673902893066406,
52
  "step": 30
53
  },
54
  {
55
  "epoch": 0.011666666666666667,
56
+ "grad_norm": 0.21875,
57
+ "learning_rate": 0.0005,
58
+ "loss": 2.686188316345215,
59
  "step": 35
60
  },
61
  {
62
  "epoch": 0.013333333333333334,
63
+ "grad_norm": 0.087890625,
64
+ "learning_rate": 0.0005,
65
+ "loss": 2.6556489944458006,
66
  "step": 40
67
  },
68
  {
69
  "epoch": 0.015,
70
+ "grad_norm": 0.119140625,
71
+ "learning_rate": 0.0005,
72
+ "loss": 2.66821346282959,
73
  "step": 45
74
  },
75
  {
76
  "epoch": 0.016666666666666666,
77
+ "grad_norm": 0.095703125,
78
+ "learning_rate": 0.0005,
79
+ "loss": 2.6468021392822267,
80
  "step": 50
81
  },
82
  {
83
  "epoch": 0.016666666666666666,
84
+ "eval_loss": 3.0268919467926025,
85
+ "eval_runtime": 4.7731,
86
+ "eval_samples_per_second": 31.217,
87
+ "eval_steps_per_second": 3.562,
88
  "step": 50
89
  },
90
  {
91
  "epoch": 0.018333333333333333,
92
+ "grad_norm": 139.0,
93
+ "learning_rate": 0.0005,
94
+ "loss": 2.6445695877075197,
95
  "step": 55
96
  },
97
  {
98
  "epoch": 0.02,
99
+ "grad_norm": 0.65234375,
100
+ "learning_rate": 0.0005,
101
+ "loss": 2.645796775817871,
102
  "step": 60
103
  },
104
  {
105
  "epoch": 0.021666666666666667,
106
+ "grad_norm": 0.07421875,
107
+ "learning_rate": 0.0005,
108
+ "loss": 2.6811267852783205,
109
  "step": 65
110
  },
111
  {
112
  "epoch": 0.023333333333333334,
113
+ "grad_norm": 0.08642578125,
114
+ "learning_rate": 0.0005,
115
+ "loss": 2.6605648040771483,
116
  "step": 70
117
  },
118
  {
119
  "epoch": 0.025,
120
+ "grad_norm": 0.275390625,
121
+ "learning_rate": 0.0005,
122
+ "loss": 2.668781852722168,
123
  "step": 75
124
  },
125
  {
126
  "epoch": 0.02666666666666667,
127
+ "grad_norm": 0.0693359375,
128
+ "learning_rate": 0.0005,
129
+ "loss": 2.6526798248291015,
130
  "step": 80
131
  },
132
  {
133
  "epoch": 0.028333333333333332,
134
+ "grad_norm": 0.427734375,
135
+ "learning_rate": 0.0005,
136
+ "loss": 2.6234062194824217,
137
  "step": 85
138
  },
139
  {
140
  "epoch": 0.03,
141
+ "grad_norm": 0.08447265625,
142
+ "learning_rate": 0.0005,
143
+ "loss": 2.6748191833496096,
144
  "step": 90
145
  },
146
  {
147
  "epoch": 0.03166666666666667,
148
+ "grad_norm": 0.08544921875,
149
+ "learning_rate": 0.0005,
150
+ "loss": 2.665359878540039,
151
  "step": 95
152
  },
153
  {
154
  "epoch": 0.03333333333333333,
155
+ "grad_norm": 0.2734375,
156
+ "learning_rate": 0.0005,
157
+ "loss": 2.6298377990722654,
158
  "step": 100
159
  },
160
  {
161
  "epoch": 0.03333333333333333,
162
+ "eval_loss": 3.037663221359253,
163
+ "eval_runtime": 4.6626,
164
+ "eval_samples_per_second": 31.956,
165
+ "eval_steps_per_second": 3.646,
166
  "step": 100
167
  },
168
  {
169
  "epoch": 0.035,
170
+ "grad_norm": 0.1865234375,
171
+ "learning_rate": 0.0005,
172
+ "loss": 2.639477348327637,
173
  "step": 105
174
  },
175
  {
176
  "epoch": 0.03666666666666667,
177
+ "grad_norm": 0.1376953125,
178
+ "learning_rate": 0.0005,
179
+ "loss": 2.671689987182617,
180
  "step": 110
181
  },
182
  {
183
  "epoch": 0.03833333333333333,
184
+ "grad_norm": 0.078125,
185
+ "learning_rate": 0.0005,
186
+ "loss": 2.6607818603515625,
187
  "step": 115
188
  },
189
  {
190
  "epoch": 0.04,
191
+ "grad_norm": 0.06494140625,
192
+ "learning_rate": 0.0005,
193
+ "loss": 2.6763479232788088,
194
  "step": 120
195
  },
196
  {
197
  "epoch": 0.041666666666666664,
198
+ "grad_norm": 0.96875,
199
+ "learning_rate": 0.0005,
200
+ "loss": 2.6471065521240233,
201
  "step": 125
202
  },
203
  {
204
  "epoch": 0.043333333333333335,
205
+ "grad_norm": 0.076171875,
206
+ "learning_rate": 0.0005,
207
+ "loss": 2.643071746826172,
208
  "step": 130
209
  },
210
  {
211
  "epoch": 0.045,
212
+ "grad_norm": 0.10400390625,
213
+ "learning_rate": 0.0005,
214
+ "loss": 2.6617103576660157,
215
  "step": 135
216
  },
217
  {
218
  "epoch": 0.04666666666666667,
219
+ "grad_norm": 0.06640625,
220
+ "learning_rate": 0.0005,
221
+ "loss": 2.617737579345703,
222
  "step": 140
223
  },
224
  {
225
  "epoch": 0.04833333333333333,
226
+ "grad_norm": 0.3984375,
227
+ "learning_rate": 0.0005,
228
+ "loss": 2.7076738357543944,
229
  "step": 145
230
  },
231
  {
232
  "epoch": 0.05,
233
+ "grad_norm": 0.09375,
234
+ "learning_rate": 0.0005,
235
+ "loss": 2.6471235275268556,
236
  "step": 150
237
  },
238
  {
239
  "epoch": 0.05,
240
+ "eval_loss": 3.055941581726074,
241
+ "eval_runtime": 4.6673,
242
+ "eval_samples_per_second": 31.924,
243
+ "eval_steps_per_second": 3.642,
244
  "step": 150
245
  },
246
  {
247
  "epoch": 0.051666666666666666,
248
+ "grad_norm": 0.08642578125,
249
+ "learning_rate": 0.0005,
250
+ "loss": 2.6089550018310548,
251
  "step": 155
252
  },
253
  {
254
  "epoch": 0.05333333333333334,
255
+ "grad_norm": 0.61328125,
256
+ "learning_rate": 0.0005,
257
+ "loss": 2.6288101196289064,
258
  "step": 160
259
  },
260
  {
261
  "epoch": 0.055,
262
+ "grad_norm": 0.0703125,
263
+ "learning_rate": 0.0005,
264
+ "loss": 2.647333335876465,
265
  "step": 165
266
  },
267
  {
268
  "epoch": 0.056666666666666664,
269
+ "grad_norm": 0.06689453125,
270
+ "learning_rate": 0.0005,
271
+ "loss": 2.6732242584228514,
272
  "step": 170
273
  },
274
  {
275
  "epoch": 0.058333333333333334,
276
+ "grad_norm": 0.060791015625,
277
+ "learning_rate": 0.0005,
278
+ "loss": 2.6399274826049806,
279
  "step": 175
280
  },
281
  {
282
  "epoch": 0.06,
283
+ "grad_norm": 0.059326171875,
284
+ "learning_rate": 0.0005,
285
+ "loss": 2.6433204650878905,
286
  "step": 180
287
  },
288
  {
289
  "epoch": 0.06166666666666667,
290
+ "grad_norm": 0.05712890625,
291
+ "learning_rate": 0.0005,
292
+ "loss": 2.6490320205688476,
293
  "step": 185
294
  },
295
  {
296
  "epoch": 0.06333333333333334,
297
+ "grad_norm": 0.0830078125,
298
+ "learning_rate": 0.0005,
299
+ "loss": 2.6651992797851562,
300
  "step": 190
301
  },
302
  {
303
  "epoch": 0.065,
304
+ "grad_norm": 0.0712890625,
305
+ "learning_rate": 0.0005,
306
+ "loss": 2.6375024795532225,
307
  "step": 195
308
  },
309
  {
310
  "epoch": 0.06666666666666667,
311
+ "grad_norm": 0.07763671875,
312
+ "learning_rate": 0.0005,
313
+ "loss": 2.616963768005371,
314
  "step": 200
315
  },
316
  {
317
  "epoch": 0.06666666666666667,
318
+ "eval_loss": 3.0367250442504883,
319
+ "eval_runtime": 4.6782,
320
+ "eval_samples_per_second": 31.85,
321
+ "eval_steps_per_second": 3.634,
322
  "step": 200
323
  }
324
  ],
 
339
  "attributes": {}
340
  }
341
  },
342
+ "total_flos": 6.52309029715968e+16,
343
+ "train_batch_size": 9,
344
  "trial_name": null,
345
  "trial_params": null
346
  }
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2fa13de9d6de96ae7af42b5d8ce1dda570123e506d560d8d7900c3522fff7dfe
3
  size 5265
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8120bbf745be8a2e230b2b8e6929b1ee58e282eec5683185cd975c85928dd837
3
  size 5265