devoppro commited on
Commit
6608932
·
verified ·
1 Parent(s): c09d1d2

Training in progress, step 400, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ba8406c53f83c09c1ffeabb8ce1224cbfb422c48a8d00576b75192664a252c36
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:81de9b280c9780cf3feaf944ed951aca1ee11fa1f451d8fe3983a73cd5acf1ff
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8a0eee39df69ccbd8f983b1f8a785ab0e689171b4e4bf4b721400c5924aa6fe6
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af4dab1bd40e26061d7373627d7696e6ebcd35c8a058d709a83490dc1881eb7c
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b31b319d83cac2dd433a790a8abad45ac5c816140a98898786a198f16bd883cd
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:74d64db17f5541ae52946e5b4e988b9c8496b896a7fd08f53e3196d74b6433cb
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9cf04c7f94b59cd606c42cf0fc828e78ab899584bb7bbb628645ada4c3c807cf
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be625e9e3b7c2d0674d4d1ce1044104bf1d523b3db3eb682bb066436988eb458
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.006,
6
  "eval_steps": 500,
7
- "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -218,6 +218,76 @@
218
  "learning_rate": 0.00029880360721442885,
219
  "loss": 10.386916351318359,
220
  "step": 300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
221
  }
222
  ],
223
  "logging_steps": 10,
@@ -237,7 +307,7 @@
237
  "attributes": {}
238
  }
239
  },
240
- "total_flos": 5668312449024000.0,
241
  "train_batch_size": 2,
242
  "trial_name": null,
243
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.008,
6
  "eval_steps": 500,
7
+ "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
218
  "learning_rate": 0.00029880360721442885,
219
  "loss": 10.386916351318359,
220
  "step": 300
221
+ },
222
+ {
223
+ "epoch": 0.0062,
224
+ "grad_norm": 2.5315372943878174,
225
+ "learning_rate": 0.00029874348697394784,
226
+ "loss": 8.981736755371093,
227
+ "step": 310
228
+ },
229
+ {
230
+ "epoch": 0.0064,
231
+ "grad_norm": 1.1698753833770752,
232
+ "learning_rate": 0.0002986833667334669,
233
+ "loss": 8.788543701171875,
234
+ "step": 320
235
+ },
236
+ {
237
+ "epoch": 0.0066,
238
+ "grad_norm": 1.353128433227539,
239
+ "learning_rate": 0.00029862324649298594,
240
+ "loss": 9.83536148071289,
241
+ "step": 330
242
+ },
243
+ {
244
+ "epoch": 0.0068,
245
+ "grad_norm": 1.1758288145065308,
246
+ "learning_rate": 0.000298563126252505,
247
+ "loss": 11.545357513427735,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 0.007,
252
+ "grad_norm": 1.2862794399261475,
253
+ "learning_rate": 0.00029850300601202403,
254
+ "loss": 9.087370300292969,
255
+ "step": 350
256
+ },
257
+ {
258
+ "epoch": 0.0072,
259
+ "grad_norm": 1.449062466621399,
260
+ "learning_rate": 0.0002984428857715431,
261
+ "loss": 11.168739318847656,
262
+ "step": 360
263
+ },
264
+ {
265
+ "epoch": 0.0074,
266
+ "grad_norm": 0.5698562264442444,
267
+ "learning_rate": 0.00029838276553106213,
268
+ "loss": 7.454578399658203,
269
+ "step": 370
270
+ },
271
+ {
272
+ "epoch": 0.0076,
273
+ "grad_norm": 0.9205979108810425,
274
+ "learning_rate": 0.0002983226452905811,
275
+ "loss": 11.321672058105468,
276
+ "step": 380
277
+ },
278
+ {
279
+ "epoch": 0.0078,
280
+ "grad_norm": 0.8094843626022339,
281
+ "learning_rate": 0.00029826252505010017,
282
+ "loss": 9.661857604980469,
283
+ "step": 390
284
+ },
285
+ {
286
+ "epoch": 0.008,
287
+ "grad_norm": 0.9517568945884705,
288
+ "learning_rate": 0.0002982024048096192,
289
+ "loss": 7.4337646484375,
290
+ "step": 400
291
  }
292
  ],
293
  "logging_steps": 10,
 
307
  "attributes": {}
308
  }
309
  },
310
+ "total_flos": 7557749932032000.0,
311
  "train_batch_size": 2,
312
  "trial_name": null,
313
  "trial_params": null