CodeIsAbstract commited on
Commit
23c6797
·
verified ·
1 Parent(s): e5d36ab

Training in progress, step 4500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6150dc5bd051bed762db3300592d4cc7ebe9703885d3d6313370582e89e126b1
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f76e1c71a592d741fdd40918c6891542d3837ccabc9fa8cbe1831ad73bed3377
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ced7feb6339932dd38c7dd1b41e44f03af228749aaf328c571f8b3d5e79694bc
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:79b075c21268b2a7ca8b93698d0da84575f538e338132fe643c3c6880a8693db
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:df4b1a89b85b3ec35f63283289082575fe9a15b34a21a7f947d7912552e363be
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9bbfe0b3b83ac16896ad88a72baf7b907818b1deab71bce85f11ea4f6347d4c
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c3220e96802ad35e508b642aabfd7e1cb6a8b7c1925ae81918e4ab19428a5638
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cfc4607d0eab509f8bd12c06c7cf681bedc18d3448143992c224d019b4cd4e30
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.8,
6
  "eval_steps": 100,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3128,6 +3128,396 @@
3128
  "eval_samples_per_second": 66.73,
3129
  "eval_steps_per_second": 3.807,
3130
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3131
  }
3132
  ],
3133
  "logging_steps": 10,
@@ -3147,7 +3537,7 @@
3147
  "attributes": {}
3148
  }
3149
  },
3150
- "total_flos": 1.304618059431936e+18,
3151
  "train_batch_size": 18,
3152
  "trial_name": null,
3153
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.9,
6
  "eval_steps": 100,
7
+ "global_step": 4500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3128
  "eval_samples_per_second": 66.73,
3129
  "eval_steps_per_second": 3.807,
3130
  "step": 4000
3131
+ },
3132
+ {
3133
+ "epoch": 0.802,
3134
+ "grad_norm": 0.054931640625,
3135
+ "learning_rate": 0.0003,
3136
+ "loss": 2.5756288528442384,
3137
+ "step": 4010
3138
+ },
3139
+ {
3140
+ "epoch": 0.804,
3141
+ "grad_norm": 0.5546875,
3142
+ "learning_rate": 0.0003,
3143
+ "loss": 2.591953468322754,
3144
+ "step": 4020
3145
+ },
3146
+ {
3147
+ "epoch": 0.806,
3148
+ "grad_norm": 0.05029296875,
3149
+ "learning_rate": 0.0003,
3150
+ "loss": 2.5956727981567385,
3151
+ "step": 4030
3152
+ },
3153
+ {
3154
+ "epoch": 0.808,
3155
+ "grad_norm": 0.05224609375,
3156
+ "learning_rate": 0.0003,
3157
+ "loss": 2.60211067199707,
3158
+ "step": 4040
3159
+ },
3160
+ {
3161
+ "epoch": 0.81,
3162
+ "grad_norm": 0.049072265625,
3163
+ "learning_rate": 0.0003,
3164
+ "loss": 2.5871608734130858,
3165
+ "step": 4050
3166
+ },
3167
+ {
3168
+ "epoch": 0.812,
3169
+ "grad_norm": 0.10693359375,
3170
+ "learning_rate": 0.0003,
3171
+ "loss": 2.5698373794555662,
3172
+ "step": 4060
3173
+ },
3174
+ {
3175
+ "epoch": 0.814,
3176
+ "grad_norm": 0.05810546875,
3177
+ "learning_rate": 0.0003,
3178
+ "loss": 2.6126802444458006,
3179
+ "step": 4070
3180
+ },
3181
+ {
3182
+ "epoch": 0.816,
3183
+ "grad_norm": 0.05029296875,
3184
+ "learning_rate": 0.0003,
3185
+ "loss": 2.5981868743896483,
3186
+ "step": 4080
3187
+ },
3188
+ {
3189
+ "epoch": 0.818,
3190
+ "grad_norm": 0.052978515625,
3191
+ "learning_rate": 0.0003,
3192
+ "loss": 2.580386161804199,
3193
+ "step": 4090
3194
+ },
3195
+ {
3196
+ "epoch": 0.82,
3197
+ "grad_norm": 0.07080078125,
3198
+ "learning_rate": 0.0003,
3199
+ "loss": 2.5840375900268553,
3200
+ "step": 4100
3201
+ },
3202
+ {
3203
+ "epoch": 0.82,
3204
+ "eval_loss": 3.0312201976776123,
3205
+ "eval_runtime": 4.4429,
3206
+ "eval_samples_per_second": 67.074,
3207
+ "eval_steps_per_second": 3.826,
3208
+ "step": 4100
3209
+ },
3210
+ {
3211
+ "epoch": 0.822,
3212
+ "grad_norm": 0.052978515625,
3213
+ "learning_rate": 0.0003,
3214
+ "loss": 2.6040040969848635,
3215
+ "step": 4110
3216
+ },
3217
+ {
3218
+ "epoch": 0.824,
3219
+ "grad_norm": 0.05224609375,
3220
+ "learning_rate": 0.0003,
3221
+ "loss": 2.5784271240234373,
3222
+ "step": 4120
3223
+ },
3224
+ {
3225
+ "epoch": 0.826,
3226
+ "grad_norm": 0.05078125,
3227
+ "learning_rate": 0.0003,
3228
+ "loss": 2.5992313385009767,
3229
+ "step": 4130
3230
+ },
3231
+ {
3232
+ "epoch": 0.828,
3233
+ "grad_norm": 0.62109375,
3234
+ "learning_rate": 0.0003,
3235
+ "loss": 2.607274055480957,
3236
+ "step": 4140
3237
+ },
3238
+ {
3239
+ "epoch": 0.83,
3240
+ "grad_norm": 0.05126953125,
3241
+ "learning_rate": 0.0003,
3242
+ "loss": 2.5900325775146484,
3243
+ "step": 4150
3244
+ },
3245
+ {
3246
+ "epoch": 0.832,
3247
+ "grad_norm": 0.0556640625,
3248
+ "learning_rate": 0.0003,
3249
+ "loss": 2.584280586242676,
3250
+ "step": 4160
3251
+ },
3252
+ {
3253
+ "epoch": 0.834,
3254
+ "grad_norm": 0.06103515625,
3255
+ "learning_rate": 0.0003,
3256
+ "loss": 2.578823471069336,
3257
+ "step": 4170
3258
+ },
3259
+ {
3260
+ "epoch": 0.836,
3261
+ "grad_norm": 0.05908203125,
3262
+ "learning_rate": 0.0003,
3263
+ "loss": 2.600706672668457,
3264
+ "step": 4180
3265
+ },
3266
+ {
3267
+ "epoch": 0.838,
3268
+ "grad_norm": 0.054931640625,
3269
+ "learning_rate": 0.0003,
3270
+ "loss": 2.5908893585205077,
3271
+ "step": 4190
3272
+ },
3273
+ {
3274
+ "epoch": 0.84,
3275
+ "grad_norm": 0.08740234375,
3276
+ "learning_rate": 0.0003,
3277
+ "loss": 2.6078845977783205,
3278
+ "step": 4200
3279
+ },
3280
+ {
3281
+ "epoch": 0.84,
3282
+ "eval_loss": 3.0311102867126465,
3283
+ "eval_runtime": 4.4714,
3284
+ "eval_samples_per_second": 66.645,
3285
+ "eval_steps_per_second": 3.802,
3286
+ "step": 4200
3287
+ },
3288
+ {
3289
+ "epoch": 0.842,
3290
+ "grad_norm": 0.0556640625,
3291
+ "learning_rate": 0.0003,
3292
+ "loss": 2.630947303771973,
3293
+ "step": 4210
3294
+ },
3295
+ {
3296
+ "epoch": 0.844,
3297
+ "grad_norm": 0.09423828125,
3298
+ "learning_rate": 0.0003,
3299
+ "loss": 2.623428726196289,
3300
+ "step": 4220
3301
+ },
3302
+ {
3303
+ "epoch": 0.846,
3304
+ "grad_norm": 0.111328125,
3305
+ "learning_rate": 0.0003,
3306
+ "loss": 2.5977907180786133,
3307
+ "step": 4230
3308
+ },
3309
+ {
3310
+ "epoch": 0.848,
3311
+ "grad_norm": 0.455078125,
3312
+ "learning_rate": 0.0003,
3313
+ "loss": 2.6008005142211914,
3314
+ "step": 4240
3315
+ },
3316
+ {
3317
+ "epoch": 0.85,
3318
+ "grad_norm": 1.6015625,
3319
+ "learning_rate": 0.0003,
3320
+ "loss": 2.5678030014038087,
3321
+ "step": 4250
3322
+ },
3323
+ {
3324
+ "epoch": 0.852,
3325
+ "grad_norm": 0.052490234375,
3326
+ "learning_rate": 0.0003,
3327
+ "loss": 2.613857650756836,
3328
+ "step": 4260
3329
+ },
3330
+ {
3331
+ "epoch": 0.854,
3332
+ "grad_norm": 0.08349609375,
3333
+ "learning_rate": 0.0003,
3334
+ "loss": 2.579903221130371,
3335
+ "step": 4270
3336
+ },
3337
+ {
3338
+ "epoch": 0.856,
3339
+ "grad_norm": 0.049560546875,
3340
+ "learning_rate": 0.0003,
3341
+ "loss": 2.579732131958008,
3342
+ "step": 4280
3343
+ },
3344
+ {
3345
+ "epoch": 0.858,
3346
+ "grad_norm": 0.181640625,
3347
+ "learning_rate": 0.0003,
3348
+ "loss": 2.5955440521240236,
3349
+ "step": 4290
3350
+ },
3351
+ {
3352
+ "epoch": 0.86,
3353
+ "grad_norm": 0.05712890625,
3354
+ "learning_rate": 0.0003,
3355
+ "loss": 2.592930793762207,
3356
+ "step": 4300
3357
+ },
3358
+ {
3359
+ "epoch": 0.86,
3360
+ "eval_loss": 3.031736373901367,
3361
+ "eval_runtime": 4.4299,
3362
+ "eval_samples_per_second": 67.27,
3363
+ "eval_steps_per_second": 3.838,
3364
+ "step": 4300
3365
+ },
3366
+ {
3367
+ "epoch": 0.862,
3368
+ "grad_norm": 0.05419921875,
3369
+ "learning_rate": 0.0003,
3370
+ "loss": 2.592413902282715,
3371
+ "step": 4310
3372
+ },
3373
+ {
3374
+ "epoch": 0.864,
3375
+ "grad_norm": 0.10986328125,
3376
+ "learning_rate": 0.0003,
3377
+ "loss": 2.569998931884766,
3378
+ "step": 4320
3379
+ },
3380
+ {
3381
+ "epoch": 0.866,
3382
+ "grad_norm": 0.048583984375,
3383
+ "learning_rate": 0.0003,
3384
+ "loss": 2.5922664642333983,
3385
+ "step": 4330
3386
+ },
3387
+ {
3388
+ "epoch": 0.868,
3389
+ "grad_norm": 0.052734375,
3390
+ "learning_rate": 0.0003,
3391
+ "loss": 2.6012556076049806,
3392
+ "step": 4340
3393
+ },
3394
+ {
3395
+ "epoch": 0.87,
3396
+ "grad_norm": 0.07177734375,
3397
+ "learning_rate": 0.0003,
3398
+ "loss": 2.6031923294067383,
3399
+ "step": 4350
3400
+ },
3401
+ {
3402
+ "epoch": 0.872,
3403
+ "grad_norm": 0.07763671875,
3404
+ "learning_rate": 0.0003,
3405
+ "loss": 2.59698543548584,
3406
+ "step": 4360
3407
+ },
3408
+ {
3409
+ "epoch": 0.874,
3410
+ "grad_norm": 0.051513671875,
3411
+ "learning_rate": 0.0003,
3412
+ "loss": 2.6096282958984376,
3413
+ "step": 4370
3414
+ },
3415
+ {
3416
+ "epoch": 0.876,
3417
+ "grad_norm": 0.052734375,
3418
+ "learning_rate": 0.0003,
3419
+ "loss": 2.6013412475585938,
3420
+ "step": 4380
3421
+ },
3422
+ {
3423
+ "epoch": 0.878,
3424
+ "grad_norm": 0.051025390625,
3425
+ "learning_rate": 0.0003,
3426
+ "loss": 2.570247268676758,
3427
+ "step": 4390
3428
+ },
3429
+ {
3430
+ "epoch": 0.88,
3431
+ "grad_norm": 0.19140625,
3432
+ "learning_rate": 0.0003,
3433
+ "loss": 2.6004173278808596,
3434
+ "step": 4400
3435
+ },
3436
+ {
3437
+ "epoch": 0.88,
3438
+ "eval_loss": 3.032118320465088,
3439
+ "eval_runtime": 4.4552,
3440
+ "eval_samples_per_second": 66.887,
3441
+ "eval_steps_per_second": 3.816,
3442
+ "step": 4400
3443
+ },
3444
+ {
3445
+ "epoch": 0.882,
3446
+ "grad_norm": 0.050048828125,
3447
+ "learning_rate": 0.0003,
3448
+ "loss": 2.60999755859375,
3449
+ "step": 4410
3450
+ },
3451
+ {
3452
+ "epoch": 0.884,
3453
+ "grad_norm": 0.1044921875,
3454
+ "learning_rate": 0.0003,
3455
+ "loss": 2.5999217987060548,
3456
+ "step": 4420
3457
+ },
3458
+ {
3459
+ "epoch": 0.886,
3460
+ "grad_norm": 0.255859375,
3461
+ "learning_rate": 0.0003,
3462
+ "loss": 2.57828369140625,
3463
+ "step": 4430
3464
+ },
3465
+ {
3466
+ "epoch": 0.888,
3467
+ "grad_norm": 0.0556640625,
3468
+ "learning_rate": 0.0003,
3469
+ "loss": 2.5876300811767576,
3470
+ "step": 4440
3471
+ },
3472
+ {
3473
+ "epoch": 0.89,
3474
+ "grad_norm": 0.052490234375,
3475
+ "learning_rate": 0.0003,
3476
+ "loss": 2.5927480697631835,
3477
+ "step": 4450
3478
+ },
3479
+ {
3480
+ "epoch": 0.892,
3481
+ "grad_norm": 0.076171875,
3482
+ "learning_rate": 0.0003,
3483
+ "loss": 2.6045318603515626,
3484
+ "step": 4460
3485
+ },
3486
+ {
3487
+ "epoch": 0.894,
3488
+ "grad_norm": 0.06201171875,
3489
+ "learning_rate": 0.0003,
3490
+ "loss": 2.585988235473633,
3491
+ "step": 4470
3492
+ },
3493
+ {
3494
+ "epoch": 0.896,
3495
+ "grad_norm": 0.050048828125,
3496
+ "learning_rate": 0.0003,
3497
+ "loss": 2.613155555725098,
3498
+ "step": 4480
3499
+ },
3500
+ {
3501
+ "epoch": 0.898,
3502
+ "grad_norm": 4.125,
3503
+ "learning_rate": 0.0003,
3504
+ "loss": 2.592078971862793,
3505
+ "step": 4490
3506
+ },
3507
+ {
3508
+ "epoch": 0.9,
3509
+ "grad_norm": 0.048095703125,
3510
+ "learning_rate": 0.0003,
3511
+ "loss": 2.591631317138672,
3512
+ "step": 4500
3513
+ },
3514
+ {
3515
+ "epoch": 0.9,
3516
+ "eval_loss": 3.028334617614746,
3517
+ "eval_runtime": 4.4441,
3518
+ "eval_samples_per_second": 67.055,
3519
+ "eval_steps_per_second": 3.825,
3520
+ "step": 4500
3521
  }
3522
  ],
3523
  "logging_steps": 10,
 
3537
  "attributes": {}
3538
  }
3539
  },
3540
+ "total_flos": 1.467695316860928e+18,
3541
  "train_batch_size": 18,
3542
  "trial_name": null,
3543
  "trial_params": null