CodeIsAbstract commited on
Commit
6fbd661
·
verified ·
1 Parent(s): 949cacb

Training in progress, step 44000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8302b42850bb9a971a92bf21001a99a1b108886f1a1c6d46589b31593094b2b
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0302f359737a0861a08d5300ffbacaf83fa3a8be231fe4f4d1d48ec4bc868703
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:56878bd403ec25d1287ace7a734c802826405ce51180d174a63debd4614a56de
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0c65dd0b573072496bd5c019f7a3adc94aeef2c65e8f10a440ac060ae875a09
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ae2914830368f46e9c6d2b2e5cbe7fbed0fa5772149d7c8cbec7c8589a46483c
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ff17261e7b5e5b975d26b5586b146fae1f6df93e96d395e1220068624854c3c
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a009a4564c35728dfec89ffdf58a6a6918940a3d67f64f03a0ddf9185dba8b9e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:52822ae525808556d105455f8693d96c34e031a74ca2ba13cffb17b581b87cf5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.5194805194805194,
6
  "eval_steps": 1000,
7
- "global_step": 40000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3128,6 +3128,318 @@
3128
  "eval_samples_per_second": 36.502,
3129
  "eval_steps_per_second": 9.125,
3130
  "step": 40000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3131
  }
3132
  ],
3133
  "logging_steps": 100,
@@ -3147,7 +3459,7 @@
3147
  "attributes": {}
3148
  }
3149
  },
3150
- "total_flos": 1.1907725524992e+18,
3151
  "train_batch_size": 22,
3152
  "trial_name": null,
3153
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.5714285714285714,
6
  "eval_steps": 1000,
7
+ "global_step": 44000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3128
  "eval_samples_per_second": 36.502,
3129
  "eval_steps_per_second": 9.125,
3130
  "step": 40000
3131
+ },
3132
+ {
3133
+ "epoch": 0.5207792207792208,
3134
+ "grad_norm": 0.22620290517807007,
3135
+ "learning_rate": 0.0007151438460054607,
3136
+ "loss": 2.7895,
3137
+ "step": 40100
3138
+ },
3139
+ {
3140
+ "epoch": 0.522077922077922,
3141
+ "grad_norm": 0.21930238604545593,
3142
+ "learning_rate": 0.000713841489430064,
3143
+ "loss": 2.7593,
3144
+ "step": 40200
3145
+ },
3146
+ {
3147
+ "epoch": 0.5233766233766234,
3148
+ "grad_norm": 0.232050359249115,
3149
+ "learning_rate": 0.0007125373548334219,
3150
+ "loss": 2.7341,
3151
+ "step": 40300
3152
+ },
3153
+ {
3154
+ "epoch": 0.5246753246753246,
3155
+ "grad_norm": 0.22093582153320312,
3156
+ "learning_rate": 0.0007112314530589825,
3157
+ "loss": 2.7415,
3158
+ "step": 40400
3159
+ },
3160
+ {
3161
+ "epoch": 0.525974025974026,
3162
+ "grad_norm": 0.21834588050842285,
3163
+ "learning_rate": 0.0007099237949648867,
3164
+ "loss": 2.7734,
3165
+ "step": 40500
3166
+ },
3167
+ {
3168
+ "epoch": 0.5272727272727272,
3169
+ "grad_norm": 0.2412509173154831,
3170
+ "learning_rate": 0.0007086143914238792,
3171
+ "loss": 2.7344,
3172
+ "step": 40600
3173
+ },
3174
+ {
3175
+ "epoch": 0.5285714285714286,
3176
+ "grad_norm": 0.2703639268875122,
3177
+ "learning_rate": 0.0007073032533232172,
3178
+ "loss": 2.7758,
3179
+ "step": 40700
3180
+ },
3181
+ {
3182
+ "epoch": 0.5298701298701298,
3183
+ "grad_norm": 0.26110807061195374,
3184
+ "learning_rate": 0.0007059903915645802,
3185
+ "loss": 2.7708,
3186
+ "step": 40800
3187
+ },
3188
+ {
3189
+ "epoch": 0.5311688311688312,
3190
+ "grad_norm": 0.2253560870885849,
3191
+ "learning_rate": 0.0007046758170639795,
3192
+ "loss": 2.7728,
3193
+ "step": 40900
3194
+ },
3195
+ {
3196
+ "epoch": 0.5324675324675324,
3197
+ "grad_norm": 0.2106214314699173,
3198
+ "learning_rate": 0.0007033595407516674,
3199
+ "loss": 2.7487,
3200
+ "step": 41000
3201
+ },
3202
+ {
3203
+ "epoch": 0.5324675324675324,
3204
+ "eval_loss": 3.1333227157592773,
3205
+ "eval_runtime": 15.8871,
3206
+ "eval_samples_per_second": 36.256,
3207
+ "eval_steps_per_second": 9.064,
3208
+ "step": 41000
3209
+ },
3210
+ {
3211
+ "epoch": 0.5337662337662338,
3212
+ "grad_norm": 0.21964064240455627,
3213
+ "learning_rate": 0.0007020415735720458,
3214
+ "loss": 2.7644,
3215
+ "step": 41100
3216
+ },
3217
+ {
3218
+ "epoch": 0.535064935064935,
3219
+ "grad_norm": 0.2143610268831253,
3220
+ "learning_rate": 0.0007007219264835758,
3221
+ "loss": 2.7476,
3222
+ "step": 41200
3223
+ },
3224
+ {
3225
+ "epoch": 0.5363636363636364,
3226
+ "grad_norm": 0.3317408263683319,
3227
+ "learning_rate": 0.0006994006104586865,
3228
+ "loss": 2.7193,
3229
+ "step": 41300
3230
+ },
3231
+ {
3232
+ "epoch": 0.5376623376623376,
3233
+ "grad_norm": 0.2196430265903473,
3234
+ "learning_rate": 0.0006980776364836834,
3235
+ "loss": 2.7699,
3236
+ "step": 41400
3237
+ },
3238
+ {
3239
+ "epoch": 0.538961038961039,
3240
+ "grad_norm": 0.2349650114774704,
3241
+ "learning_rate": 0.0006967530155586577,
3242
+ "loss": 2.7812,
3243
+ "step": 41500
3244
+ },
3245
+ {
3246
+ "epoch": 0.5402597402597402,
3247
+ "grad_norm": 0.20911413431167603,
3248
+ "learning_rate": 0.0006954267586973939,
3249
+ "loss": 2.7562,
3250
+ "step": 41600
3251
+ },
3252
+ {
3253
+ "epoch": 0.5415584415584416,
3254
+ "grad_norm": 0.2115071713924408,
3255
+ "learning_rate": 0.0006940988769272794,
3256
+ "loss": 2.7577,
3257
+ "step": 41700
3258
+ },
3259
+ {
3260
+ "epoch": 0.5428571428571428,
3261
+ "grad_norm": 0.24407769739627838,
3262
+ "learning_rate": 0.0006927693812892116,
3263
+ "loss": 2.7342,
3264
+ "step": 41800
3265
+ },
3266
+ {
3267
+ "epoch": 0.5441558441558442,
3268
+ "grad_norm": 0.21947546303272247,
3269
+ "learning_rate": 0.0006914382828375068,
3270
+ "loss": 2.7554,
3271
+ "step": 41900
3272
+ },
3273
+ {
3274
+ "epoch": 0.5454545454545454,
3275
+ "grad_norm": 0.2191598266363144,
3276
+ "learning_rate": 0.000690105592639809,
3277
+ "loss": 2.7256,
3278
+ "step": 42000
3279
+ },
3280
+ {
3281
+ "epoch": 0.5454545454545454,
3282
+ "eval_loss": 3.1299455165863037,
3283
+ "eval_runtime": 15.1552,
3284
+ "eval_samples_per_second": 38.007,
3285
+ "eval_steps_per_second": 9.502,
3286
+ "step": 42000
3287
+ },
3288
+ {
3289
+ "epoch": 0.5467532467532468,
3290
+ "grad_norm": 0.22920465469360352,
3291
+ "learning_rate": 0.0006887713217769954,
3292
+ "loss": 2.7328,
3293
+ "step": 42100
3294
+ },
3295
+ {
3296
+ "epoch": 0.548051948051948,
3297
+ "grad_norm": 0.23324090242385864,
3298
+ "learning_rate": 0.0006874354813430874,
3299
+ "loss": 2.7545,
3300
+ "step": 42200
3301
+ },
3302
+ {
3303
+ "epoch": 0.5493506493506494,
3304
+ "grad_norm": 0.21898794174194336,
3305
+ "learning_rate": 0.0006860980824451563,
3306
+ "loss": 2.7611,
3307
+ "step": 42300
3308
+ },
3309
+ {
3310
+ "epoch": 0.5506493506493506,
3311
+ "grad_norm": 0.21969111263751984,
3312
+ "learning_rate": 0.0006847591362032313,
3313
+ "loss": 2.7413,
3314
+ "step": 42400
3315
+ },
3316
+ {
3317
+ "epoch": 0.551948051948052,
3318
+ "grad_norm": 0.24275629222393036,
3319
+ "learning_rate": 0.0006834186537502076,
3320
+ "loss": 2.7421,
3321
+ "step": 42500
3322
+ },
3323
+ {
3324
+ "epoch": 0.5532467532467532,
3325
+ "grad_norm": 0.20918132364749908,
3326
+ "learning_rate": 0.0006820766462317534,
3327
+ "loss": 2.73,
3328
+ "step": 42600
3329
+ },
3330
+ {
3331
+ "epoch": 0.5545454545454546,
3332
+ "grad_norm": 0.2445412427186966,
3333
+ "learning_rate": 0.0006807331248062174,
3334
+ "loss": 2.7811,
3335
+ "step": 42700
3336
+ },
3337
+ {
3338
+ "epoch": 0.5558441558441558,
3339
+ "grad_norm": 0.24175776541233063,
3340
+ "learning_rate": 0.0006793881006445354,
3341
+ "loss": 2.7529,
3342
+ "step": 42800
3343
+ },
3344
+ {
3345
+ "epoch": 0.5571428571428572,
3346
+ "grad_norm": 0.234651118516922,
3347
+ "learning_rate": 0.0006780415849301388,
3348
+ "loss": 2.7592,
3349
+ "step": 42900
3350
+ },
3351
+ {
3352
+ "epoch": 0.5584415584415584,
3353
+ "grad_norm": 0.24966347217559814,
3354
+ "learning_rate": 0.0006766935888588599,
3355
+ "loss": 2.7314,
3356
+ "step": 43000
3357
+ },
3358
+ {
3359
+ "epoch": 0.5584415584415584,
3360
+ "eval_loss": 3.1258981227874756,
3361
+ "eval_runtime": 15.1058,
3362
+ "eval_samples_per_second": 38.131,
3363
+ "eval_steps_per_second": 9.533,
3364
+ "step": 43000
3365
+ },
3366
+ {
3367
+ "epoch": 0.5597402597402598,
3368
+ "grad_norm": 0.30740296840667725,
3369
+ "learning_rate": 0.0006753441236388405,
3370
+ "loss": 2.7467,
3371
+ "step": 43100
3372
+ },
3373
+ {
3374
+ "epoch": 0.561038961038961,
3375
+ "grad_norm": 0.2325826734304428,
3376
+ "learning_rate": 0.0006739932004904373,
3377
+ "loss": 2.7428,
3378
+ "step": 43200
3379
+ },
3380
+ {
3381
+ "epoch": 0.5623376623376624,
3382
+ "grad_norm": 0.2326454520225525,
3383
+ "learning_rate": 0.0006726408306461294,
3384
+ "loss": 2.7515,
3385
+ "step": 43300
3386
+ },
3387
+ {
3388
+ "epoch": 0.5636363636363636,
3389
+ "grad_norm": 0.24041685461997986,
3390
+ "learning_rate": 0.000671287025350425,
3391
+ "loss": 2.7339,
3392
+ "step": 43400
3393
+ },
3394
+ {
3395
+ "epoch": 0.564935064935065,
3396
+ "grad_norm": 0.21853885054588318,
3397
+ "learning_rate": 0.0006699317958597668,
3398
+ "loss": 2.7188,
3399
+ "step": 43500
3400
+ },
3401
+ {
3402
+ "epoch": 0.5662337662337662,
3403
+ "grad_norm": 0.32151710987091064,
3404
+ "learning_rate": 0.00066857515344244,
3405
+ "loss": 2.7317,
3406
+ "step": 43600
3407
+ },
3408
+ {
3409
+ "epoch": 0.5675324675324676,
3410
+ "grad_norm": 0.2511979639530182,
3411
+ "learning_rate": 0.0006672171093784773,
3412
+ "loss": 2.7714,
3413
+ "step": 43700
3414
+ },
3415
+ {
3416
+ "epoch": 0.5688311688311688,
3417
+ "grad_norm": 0.2406754046678543,
3418
+ "learning_rate": 0.0006658576749595663,
3419
+ "loss": 2.7762,
3420
+ "step": 43800
3421
+ },
3422
+ {
3423
+ "epoch": 0.5701298701298702,
3424
+ "grad_norm": 0.23970568180084229,
3425
+ "learning_rate": 0.0006644968614889538,
3426
+ "loss": 2.7163,
3427
+ "step": 43900
3428
+ },
3429
+ {
3430
+ "epoch": 0.5714285714285714,
3431
+ "grad_norm": 0.23587392270565033,
3432
+ "learning_rate": 0.0006631346802813543,
3433
+ "loss": 2.7272,
3434
+ "step": 44000
3435
+ },
3436
+ {
3437
+ "epoch": 0.5714285714285714,
3438
+ "eval_loss": 3.120995044708252,
3439
+ "eval_runtime": 15.2013,
3440
+ "eval_samples_per_second": 37.892,
3441
+ "eval_steps_per_second": 9.473,
3442
+ "step": 44000
3443
  }
3444
  ],
3445
  "logging_steps": 100,
 
3459
  "attributes": {}
3460
  }
3461
  },
3462
+ "total_flos": 1.30984980774912e+18,
3463
  "train_batch_size": 22,
3464
  "trial_name": null,
3465
  "trial_params": null