CodeIsAbstract commited on
Commit
81a9bd4
·
verified ·
1 Parent(s): aedb18c

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fd67d7aff17d4c04f065891a8ceba243346c4e36f53d807997e52209ae039ec0
3
  size 1738460416
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b0f05902edbab8c2047e2cd165f535365973f4882aa48d72f02b579256828033
3
  size 1738460416
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:48998fa4e64b0552833c6886caf08be135639f3a2f2eedc6a2ff97fd05414a26
3
  size 3477327340
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b51c46bae4daf9e6a9e338f80a67d2671a57ba0289d103540996cd83f303338c
3
  size 3477327340
last-checkpoint/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c213a373f3e5d95993ad095a3790a902d821a1b4b93a10cc7d382c8726fcb9d
3
+ size 15429
last-checkpoint/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8fb125336725f7741cb4daa1e3d06e225bbacfde8d41c4dcabb6762c222e62c6
3
+ size 15429
last-checkpoint/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:153c865f77c7129ba565bded50f334683d51c80f20e3cfec39e62f8737b86f0d
3
+ size 15429
last-checkpoint/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d017ce00fcebac7edc058ddd138f194eb0340f2d8ad0879bdab08f922ed0846e
3
+ size 15429
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d5638739d9917cfd5ef0eac6652b3a82be017aee5432b6e7503833c5022688e4
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:043e3a8e2a2579bce7c021fb61a180d705f6161a67c5cea201876b0db210edab
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,175 +2,315 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.01,
6
- "eval_steps": 50,
7
- "global_step": 100,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0005,
14
- "grad_norm": 6.46875,
15
- "learning_rate": 2e-05,
16
- "loss": 12.1248,
17
  "step": 5
18
  },
19
  {
20
- "epoch": 0.001,
21
- "grad_norm": 3.328125,
22
- "learning_rate": 4.5e-05,
23
- "loss": 11.2119,
24
  "step": 10
25
  },
26
  {
27
- "epoch": 0.0015,
28
- "grad_norm": 1.890625,
29
- "learning_rate": 5e-05,
30
- "loss": 10.2505,
31
  "step": 15
32
  },
33
  {
34
- "epoch": 0.002,
35
- "grad_norm": 1.9375,
36
- "learning_rate": 5e-05,
37
- "loss": 9.8307,
38
  "step": 20
39
  },
40
  {
41
- "epoch": 0.0025,
42
- "grad_norm": 1.8046875,
43
- "learning_rate": 5e-05,
44
- "loss": 9.5638,
45
  "step": 25
46
  },
47
  {
48
- "epoch": 0.003,
49
- "grad_norm": 1.546875,
50
- "learning_rate": 5e-05,
51
- "loss": 9.3031,
52
  "step": 30
53
  },
54
  {
55
- "epoch": 0.0035,
56
- "grad_norm": 1.4375,
57
- "learning_rate": 5e-05,
58
- "loss": 8.97,
59
  "step": 35
60
  },
61
  {
62
- "epoch": 0.004,
63
- "grad_norm": 1.28125,
64
- "learning_rate": 5e-05,
65
- "loss": 8.7355,
66
  "step": 40
67
  },
68
  {
69
- "epoch": 0.0045,
70
- "grad_norm": 1.296875,
71
- "learning_rate": 5e-05,
72
- "loss": 8.4612,
73
  "step": 45
74
  },
75
  {
76
- "epoch": 0.005,
77
- "grad_norm": 1.1015625,
78
- "learning_rate": 5e-05,
79
- "loss": 8.2349,
80
- "step": 50
81
- },
82
- {
83
- "epoch": 0.005,
84
- "eval_loss": 8.104817390441895,
85
- "eval_runtime": 18.203,
86
- "eval_samples_per_second": 14.778,
87
- "eval_steps_per_second": 1.483,
88
  "step": 50
89
  },
90
  {
91
- "epoch": 0.0055,
92
- "grad_norm": 1.0,
93
- "learning_rate": 5e-05,
94
- "loss": 8.0171,
95
  "step": 55
96
  },
97
  {
98
- "epoch": 0.006,
99
- "grad_norm": 0.94140625,
100
- "learning_rate": 5e-05,
101
- "loss": 7.8534,
102
  "step": 60
103
  },
104
  {
105
- "epoch": 0.0065,
106
- "grad_norm": 0.765625,
107
- "learning_rate": 5e-05,
108
- "loss": 7.7173,
109
  "step": 65
110
  },
111
  {
112
- "epoch": 0.007,
113
- "grad_norm": 1.046875,
114
- "learning_rate": 5e-05,
115
- "loss": 7.535,
116
  "step": 70
117
  },
118
  {
119
- "epoch": 0.0075,
120
- "grad_norm": 1.0703125,
121
- "learning_rate": 5e-05,
122
- "loss": 7.4876,
123
  "step": 75
124
  },
125
  {
126
- "epoch": 0.008,
127
- "grad_norm": 0.98046875,
128
- "learning_rate": 5e-05,
129
- "loss": 7.4177,
130
  "step": 80
131
  },
132
  {
133
- "epoch": 0.0085,
134
- "grad_norm": 0.9296875,
135
- "learning_rate": 5e-05,
136
- "loss": 7.3471,
137
  "step": 85
138
  },
139
  {
140
- "epoch": 0.009,
141
- "grad_norm": 0.7421875,
142
- "learning_rate": 5e-05,
143
- "loss": 7.35,
144
  "step": 90
145
  },
146
  {
147
- "epoch": 0.0095,
148
- "grad_norm": 0.64453125,
149
- "learning_rate": 5e-05,
150
- "loss": 7.2766,
151
  "step": 95
152
  },
153
  {
154
- "epoch": 0.01,
155
- "grad_norm": 0.73046875,
156
- "learning_rate": 5e-05,
157
- "loss": 7.1746,
158
  "step": 100
159
  },
160
  {
161
- "epoch": 0.01,
162
- "eval_loss": 7.275140285491943,
163
- "eval_runtime": 17.5509,
164
- "eval_samples_per_second": 15.327,
165
- "eval_steps_per_second": 1.538,
166
  "step": 100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
167
  }
168
  ],
169
  "logging_steps": 5,
170
- "max_steps": 10000,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 9223372036854775807,
173
- "save_steps": 100,
174
  "stateful_callbacks": {
175
  "TrainerControl": {
176
  "args": {
@@ -183,7 +323,7 @@
183
  "attributes": {}
184
  }
185
  },
186
- "total_flos": 1.6351837028352e+16,
187
  "train_batch_size": 10,
188
  "trial_name": null,
189
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.13333333333333333,
6
+ "eval_steps": 100,
7
+ "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0033333333333333335,
14
+ "grad_norm": 27.25,
15
+ "learning_rate": 8.000000000000001e-06,
16
+ "loss": 48.7359,
17
  "step": 5
18
  },
19
  {
20
+ "epoch": 0.006666666666666667,
21
+ "grad_norm": 20.75,
22
+ "learning_rate": 1.8e-05,
23
+ "loss": 47.3893,
24
  "step": 10
25
  },
26
  {
27
+ "epoch": 0.01,
28
+ "grad_norm": 12.9375,
29
+ "learning_rate": 2e-05,
30
+ "loss": 44.1295,
31
  "step": 15
32
  },
33
  {
34
+ "epoch": 0.013333333333333334,
35
+ "grad_norm": 9.6875,
36
+ "learning_rate": 2e-05,
37
+ "loss": 41.9134,
38
  "step": 20
39
  },
40
  {
41
+ "epoch": 0.016666666666666666,
42
+ "grad_norm": 8.25,
43
+ "learning_rate": 2e-05,
44
+ "loss": 40.9631,
45
  "step": 25
46
  },
47
  {
48
+ "epoch": 0.02,
49
+ "grad_norm": 7.53125,
50
+ "learning_rate": 2e-05,
51
+ "loss": 40.2396,
52
  "step": 30
53
  },
54
  {
55
+ "epoch": 0.023333333333333334,
56
+ "grad_norm": 7.65625,
57
+ "learning_rate": 2e-05,
58
+ "loss": 39.7111,
59
  "step": 35
60
  },
61
  {
62
+ "epoch": 0.02666666666666667,
63
+ "grad_norm": 7.375,
64
+ "learning_rate": 2e-05,
65
+ "loss": 39.2189,
66
  "step": 40
67
  },
68
  {
69
+ "epoch": 0.03,
70
+ "grad_norm": 6.96875,
71
+ "learning_rate": 2e-05,
72
+ "loss": 38.8553,
73
  "step": 45
74
  },
75
  {
76
+ "epoch": 0.03333333333333333,
77
+ "grad_norm": 6.78125,
78
+ "learning_rate": 2e-05,
79
+ "loss": 38.5259,
 
 
 
 
 
 
 
 
80
  "step": 50
81
  },
82
  {
83
+ "epoch": 0.03666666666666667,
84
+ "grad_norm": 6.53125,
85
+ "learning_rate": 2e-05,
86
+ "loss": 38.059,
87
  "step": 55
88
  },
89
  {
90
+ "epoch": 0.04,
91
+ "grad_norm": 6.78125,
92
+ "learning_rate": 2e-05,
93
+ "loss": 37.7568,
94
  "step": 60
95
  },
96
  {
97
+ "epoch": 0.043333333333333335,
98
+ "grad_norm": 6.375,
99
+ "learning_rate": 2e-05,
100
+ "loss": 37.3479,
101
  "step": 65
102
  },
103
  {
104
+ "epoch": 0.04666666666666667,
105
+ "grad_norm": 6.4375,
106
+ "learning_rate": 2e-05,
107
+ "loss": 37.0288,
108
  "step": 70
109
  },
110
  {
111
+ "epoch": 0.05,
112
+ "grad_norm": 6.125,
113
+ "learning_rate": 2e-05,
114
+ "loss": 36.7166,
115
  "step": 75
116
  },
117
  {
118
+ "epoch": 0.05333333333333334,
119
+ "grad_norm": 6.0625,
120
+ "learning_rate": 2e-05,
121
+ "loss": 36.362,
122
  "step": 80
123
  },
124
  {
125
+ "epoch": 0.056666666666666664,
126
+ "grad_norm": 5.875,
127
+ "learning_rate": 2e-05,
128
+ "loss": 36.1666,
129
  "step": 85
130
  },
131
  {
132
+ "epoch": 0.06,
133
+ "grad_norm": 5.9375,
134
+ "learning_rate": 2e-05,
135
+ "loss": 35.8911,
136
  "step": 90
137
  },
138
  {
139
+ "epoch": 0.06333333333333334,
140
+ "grad_norm": 5.4375,
141
+ "learning_rate": 2e-05,
142
+ "loss": 35.6416,
143
  "step": 95
144
  },
145
  {
146
+ "epoch": 0.06666666666666667,
147
+ "grad_norm": 5.71875,
148
+ "learning_rate": 2e-05,
149
+ "loss": 35.3982,
150
  "step": 100
151
  },
152
  {
153
+ "epoch": 0.06666666666666667,
154
+ "eval_loss": 8.794044494628906,
155
+ "eval_runtime": 6.9551,
156
+ "eval_samples_per_second": 38.676,
157
+ "eval_steps_per_second": 1.006,
158
  "step": 100
159
+ },
160
+ {
161
+ "epoch": 0.07,
162
+ "grad_norm": 5.5,
163
+ "learning_rate": 2e-05,
164
+ "loss": 35.1707,
165
+ "step": 105
166
+ },
167
+ {
168
+ "epoch": 0.07333333333333333,
169
+ "grad_norm": 5.875,
170
+ "learning_rate": 2e-05,
171
+ "loss": 34.8065,
172
+ "step": 110
173
+ },
174
+ {
175
+ "epoch": 0.07666666666666666,
176
+ "grad_norm": 5.09375,
177
+ "learning_rate": 2e-05,
178
+ "loss": 34.5524,
179
+ "step": 115
180
+ },
181
+ {
182
+ "epoch": 0.08,
183
+ "grad_norm": 5.9375,
184
+ "learning_rate": 2e-05,
185
+ "loss": 34.5738,
186
+ "step": 120
187
+ },
188
+ {
189
+ "epoch": 0.08333333333333333,
190
+ "grad_norm": 4.96875,
191
+ "learning_rate": 2e-05,
192
+ "loss": 34.1881,
193
+ "step": 125
194
+ },
195
+ {
196
+ "epoch": 0.08666666666666667,
197
+ "grad_norm": 4.9375,
198
+ "learning_rate": 2e-05,
199
+ "loss": 34.0342,
200
+ "step": 130
201
+ },
202
+ {
203
+ "epoch": 0.09,
204
+ "grad_norm": 4.9375,
205
+ "learning_rate": 2e-05,
206
+ "loss": 33.7996,
207
+ "step": 135
208
+ },
209
+ {
210
+ "epoch": 0.09333333333333334,
211
+ "grad_norm": 4.8125,
212
+ "learning_rate": 2e-05,
213
+ "loss": 33.5324,
214
+ "step": 140
215
+ },
216
+ {
217
+ "epoch": 0.09666666666666666,
218
+ "grad_norm": 4.90625,
219
+ "learning_rate": 2e-05,
220
+ "loss": 33.3612,
221
+ "step": 145
222
+ },
223
+ {
224
+ "epoch": 0.1,
225
+ "grad_norm": 5.03125,
226
+ "learning_rate": 2e-05,
227
+ "loss": 33.1498,
228
+ "step": 150
229
+ },
230
+ {
231
+ "epoch": 0.10333333333333333,
232
+ "grad_norm": 4.78125,
233
+ "learning_rate": 2e-05,
234
+ "loss": 33.0172,
235
+ "step": 155
236
+ },
237
+ {
238
+ "epoch": 0.10666666666666667,
239
+ "grad_norm": 4.53125,
240
+ "learning_rate": 2e-05,
241
+ "loss": 32.9073,
242
+ "step": 160
243
+ },
244
+ {
245
+ "epoch": 0.11,
246
+ "grad_norm": 4.53125,
247
+ "learning_rate": 2e-05,
248
+ "loss": 32.5479,
249
+ "step": 165
250
+ },
251
+ {
252
+ "epoch": 0.11333333333333333,
253
+ "grad_norm": 4.46875,
254
+ "learning_rate": 2e-05,
255
+ "loss": 32.4669,
256
+ "step": 170
257
+ },
258
+ {
259
+ "epoch": 0.11666666666666667,
260
+ "grad_norm": 5.8125,
261
+ "learning_rate": 2e-05,
262
+ "loss": 32.4108,
263
+ "step": 175
264
+ },
265
+ {
266
+ "epoch": 0.12,
267
+ "grad_norm": 4.5625,
268
+ "learning_rate": 2e-05,
269
+ "loss": 32.0156,
270
+ "step": 180
271
+ },
272
+ {
273
+ "epoch": 0.12333333333333334,
274
+ "grad_norm": 4.34375,
275
+ "learning_rate": 2e-05,
276
+ "loss": 32.0051,
277
+ "step": 185
278
+ },
279
+ {
280
+ "epoch": 0.12666666666666668,
281
+ "grad_norm": 4.34375,
282
+ "learning_rate": 2e-05,
283
+ "loss": 31.8771,
284
+ "step": 190
285
+ },
286
+ {
287
+ "epoch": 0.13,
288
+ "grad_norm": 4.6875,
289
+ "learning_rate": 2e-05,
290
+ "loss": 31.6117,
291
+ "step": 195
292
+ },
293
+ {
294
+ "epoch": 0.13333333333333333,
295
+ "grad_norm": 4.3125,
296
+ "learning_rate": 2e-05,
297
+ "loss": 31.5273,
298
+ "step": 200
299
+ },
300
+ {
301
+ "epoch": 0.13333333333333333,
302
+ "eval_loss": 7.888162612915039,
303
+ "eval_runtime": 6.8793,
304
+ "eval_samples_per_second": 39.103,
305
+ "eval_steps_per_second": 1.018,
306
+ "step": 200
307
  }
308
  ],
309
  "logging_steps": 5,
310
+ "max_steps": 1500,
311
  "num_input_tokens_seen": 0,
312
  "num_train_epochs": 9223372036854775807,
313
+ "save_steps": 200,
314
  "stateful_callbacks": {
315
  "TrainerControl": {
316
  "args": {
 
323
  "attributes": {}
324
  }
325
  },
326
+ "total_flos": 1.3081469656236032e+17,
327
  "train_batch_size": 10,
328
  "trial_name": null,
329
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:93663ce2422ad42e4302eebc4223254912b3e8626b07cdb378fa0ebab8884f21
3
  size 5841
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:312a0ac7d3d7dd02c5699f77d48121d29f2b0b83c1a40f2d885f5e5598f0a3af
3
  size 5841