CodeIsAbstract commited on
Commit
e6ee242
·
verified ·
1 Parent(s): 39d3d70

Training in progress, step 32000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32cc85238f9b318459c5a80206968432f8685331234104eda2016524ba706b75
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fea0a0b88a8852aec2382a6400435ae84bbfcb52cd14372fe08dab6f5a9b181
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5e66d55b95606084af7dc0fbf657d700ecafd09b008231850874bdde638deda6
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5bbb61cfa227633b44a69a3c624c1b0dbf80db3ba79898d7b764dc31de1163ee
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8553d3987609520495680f36a4682a760a418a930648ce46178a1399cde440b0
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:428aed83766f09a6a2d9ccfb86972eeda646225b3b0cd6accec4a90e4f900161
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:71903ea651461945bbccce0e330f120093c2124a33f5b93ba51f8b9e386de7f9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:35deceb007dc05c0a67e4f7d57b32921e37176fa5a3cb5bba0983b1b73cb3efc
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.36363636363636365,
6
  "eval_steps": 1000,
7
- "global_step": 28000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2192,6 +2192,318 @@
2192
  "eval_samples_per_second": 36.109,
2193
  "eval_steps_per_second": 9.027,
2194
  "step": 28000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2195
  }
2196
  ],
2197
  "logging_steps": 100,
@@ -2211,7 +2523,7 @@
2211
  "attributes": {}
2212
  }
2213
  },
2214
- "total_flos": 8.3354078674944e+17,
2215
  "train_batch_size": 22,
2216
  "trial_name": null,
2217
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4155844155844156,
6
  "eval_steps": 1000,
7
+ "global_step": 32000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2192
  "eval_samples_per_second": 36.109,
2193
  "eval_steps_per_second": 9.027,
2194
  "step": 28000
2195
+ },
2196
+ {
2197
+ "epoch": 0.36493506493506495,
2198
+ "grad_norm": 0.20181582868099213,
2199
+ "learning_rate": 0.00085546987355325,
2200
+ "loss": 2.835,
2201
+ "step": 28100
2202
+ },
2203
+ {
2204
+ "epoch": 0.36623376623376624,
2205
+ "grad_norm": 0.1909325122833252,
2206
+ "learning_rate": 0.0008544544759847112,
2207
+ "loss": 2.8291,
2208
+ "step": 28200
2209
+ },
2210
+ {
2211
+ "epoch": 0.36753246753246754,
2212
+ "grad_norm": 0.21319599449634552,
2213
+ "learning_rate": 0.000853436131244459,
2214
+ "loss": 2.8536,
2215
+ "step": 28300
2216
+ },
2217
+ {
2218
+ "epoch": 0.36883116883116884,
2219
+ "grad_norm": 0.200112447142601,
2220
+ "learning_rate": 0.0008524148477996933,
2221
+ "loss": 2.8732,
2222
+ "step": 28400
2223
+ },
2224
+ {
2225
+ "epoch": 0.37012987012987014,
2226
+ "grad_norm": 0.19826772809028625,
2227
+ "learning_rate": 0.0008513906341420478,
2228
+ "loss": 2.8208,
2229
+ "step": 28500
2230
+ },
2231
+ {
2232
+ "epoch": 0.37142857142857144,
2233
+ "grad_norm": 0.21064135432243347,
2234
+ "learning_rate": 0.0008503634987875206,
2235
+ "loss": 2.805,
2236
+ "step": 28600
2237
+ },
2238
+ {
2239
+ "epoch": 0.37272727272727274,
2240
+ "grad_norm": 0.2014266848564148,
2241
+ "learning_rate": 0.0008493334502764021,
2242
+ "loss": 2.8257,
2243
+ "step": 28700
2244
+ },
2245
+ {
2246
+ "epoch": 0.37402597402597404,
2247
+ "grad_norm": 0.21088480949401855,
2248
+ "learning_rate": 0.0008483004971732049,
2249
+ "loss": 2.8261,
2250
+ "step": 28800
2251
+ },
2252
+ {
2253
+ "epoch": 0.37532467532467534,
2254
+ "grad_norm": 0.20050457119941711,
2255
+ "learning_rate": 0.0008472646480665924,
2256
+ "loss": 2.8694,
2257
+ "step": 28900
2258
+ },
2259
+ {
2260
+ "epoch": 0.37662337662337664,
2261
+ "grad_norm": 0.21420925855636597,
2262
+ "learning_rate": 0.0008462259115693076,
2263
+ "loss": 2.8473,
2264
+ "step": 29000
2265
+ },
2266
+ {
2267
+ "epoch": 0.37662337662337664,
2268
+ "eval_loss": 3.195821523666382,
2269
+ "eval_runtime": 15.2121,
2270
+ "eval_samples_per_second": 37.865,
2271
+ "eval_steps_per_second": 9.466,
2272
+ "step": 29000
2273
+ },
2274
+ {
2275
+ "epoch": 0.37792207792207794,
2276
+ "grad_norm": 0.2159246951341629,
2277
+ "learning_rate": 0.0008451842963181003,
2278
+ "loss": 2.8032,
2279
+ "step": 29100
2280
+ },
2281
+ {
2282
+ "epoch": 0.37922077922077924,
2283
+ "grad_norm": 0.19643986225128174,
2284
+ "learning_rate": 0.0008441398109736571,
2285
+ "loss": 2.847,
2286
+ "step": 29200
2287
+ },
2288
+ {
2289
+ "epoch": 0.38051948051948054,
2290
+ "grad_norm": 0.1965179294347763,
2291
+ "learning_rate": 0.0008430924642205279,
2292
+ "loss": 2.8529,
2293
+ "step": 29300
2294
+ },
2295
+ {
2296
+ "epoch": 0.38181818181818183,
2297
+ "grad_norm": 0.21244043111801147,
2298
+ "learning_rate": 0.0008420422647670548,
2299
+ "loss": 2.8067,
2300
+ "step": 29400
2301
+ },
2302
+ {
2303
+ "epoch": 0.38311688311688313,
2304
+ "grad_norm": 0.20399615168571472,
2305
+ "learning_rate": 0.0008409892213452987,
2306
+ "loss": 2.8176,
2307
+ "step": 29500
2308
+ },
2309
+ {
2310
+ "epoch": 0.38441558441558443,
2311
+ "grad_norm": 0.20329082012176514,
2312
+ "learning_rate": 0.0008399333427109672,
2313
+ "loss": 2.8316,
2314
+ "step": 29600
2315
+ },
2316
+ {
2317
+ "epoch": 0.38571428571428573,
2318
+ "grad_norm": 0.19244763255119324,
2319
+ "learning_rate": 0.0008388746376433419,
2320
+ "loss": 2.8367,
2321
+ "step": 29700
2322
+ },
2323
+ {
2324
+ "epoch": 0.38701298701298703,
2325
+ "grad_norm": 0.18999354541301727,
2326
+ "learning_rate": 0.0008378131149452053,
2327
+ "loss": 2.8212,
2328
+ "step": 29800
2329
+ },
2330
+ {
2331
+ "epoch": 0.38831168831168833,
2332
+ "grad_norm": 0.20788761973381042,
2333
+ "learning_rate": 0.0008367487834427674,
2334
+ "loss": 2.7999,
2335
+ "step": 29900
2336
+ },
2337
+ {
2338
+ "epoch": 0.38961038961038963,
2339
+ "grad_norm": 0.21678058803081512,
2340
+ "learning_rate": 0.0008356816519855926,
2341
+ "loss": 2.81,
2342
+ "step": 30000
2343
+ },
2344
+ {
2345
+ "epoch": 0.38961038961038963,
2346
+ "eval_loss": 3.1891965866088867,
2347
+ "eval_runtime": 15.5166,
2348
+ "eval_samples_per_second": 37.122,
2349
+ "eval_steps_per_second": 9.28,
2350
+ "step": 30000
2351
+ },
2352
+ {
2353
+ "epoch": 0.39090909090909093,
2354
+ "grad_norm": 0.25863537192344666,
2355
+ "learning_rate": 0.0008346117294465258,
2356
+ "loss": 2.7919,
2357
+ "step": 30100
2358
+ },
2359
+ {
2360
+ "epoch": 0.3922077922077922,
2361
+ "grad_norm": 0.2082231491804123,
2362
+ "learning_rate": 0.0008335390247216193,
2363
+ "loss": 2.8058,
2364
+ "step": 30200
2365
+ },
2366
+ {
2367
+ "epoch": 0.3935064935064935,
2368
+ "grad_norm": 0.2484658658504486,
2369
+ "learning_rate": 0.0008324635467300578,
2370
+ "loss": 2.7913,
2371
+ "step": 30300
2372
+ },
2373
+ {
2374
+ "epoch": 0.3948051948051948,
2375
+ "grad_norm": 0.19809049367904663,
2376
+ "learning_rate": 0.000831385304414085,
2377
+ "loss": 2.8392,
2378
+ "step": 30400
2379
+ },
2380
+ {
2381
+ "epoch": 0.3961038961038961,
2382
+ "grad_norm": 0.19220763444900513,
2383
+ "learning_rate": 0.0008303043067389293,
2384
+ "loss": 2.8326,
2385
+ "step": 30500
2386
+ },
2387
+ {
2388
+ "epoch": 0.3974025974025974,
2389
+ "grad_norm": 0.19120201468467712,
2390
+ "learning_rate": 0.0008292205626927285,
2391
+ "loss": 2.8008,
2392
+ "step": 30600
2393
+ },
2394
+ {
2395
+ "epoch": 0.3987012987012987,
2396
+ "grad_norm": 0.2093675136566162,
2397
+ "learning_rate": 0.000828134081286456,
2398
+ "loss": 2.8228,
2399
+ "step": 30700
2400
+ },
2401
+ {
2402
+ "epoch": 0.4,
2403
+ "grad_norm": 0.2077227681875229,
2404
+ "learning_rate": 0.0008270448715538452,
2405
+ "loss": 2.8263,
2406
+ "step": 30800
2407
+ },
2408
+ {
2409
+ "epoch": 0.4012987012987013,
2410
+ "grad_norm": 0.2050541490316391,
2411
+ "learning_rate": 0.0008259529425513148,
2412
+ "loss": 2.8289,
2413
+ "step": 30900
2414
+ },
2415
+ {
2416
+ "epoch": 0.4025974025974026,
2417
+ "grad_norm": 0.19998504221439362,
2418
+ "learning_rate": 0.0008248583033578932,
2419
+ "loss": 2.8068,
2420
+ "step": 31000
2421
+ },
2422
+ {
2423
+ "epoch": 0.4025974025974026,
2424
+ "eval_loss": 3.1833224296569824,
2425
+ "eval_runtime": 11.0521,
2426
+ "eval_samples_per_second": 52.117,
2427
+ "eval_steps_per_second": 13.029,
2428
+ "step": 31000
2429
+ },
2430
+ {
2431
+ "epoch": 0.4038961038961039,
2432
+ "grad_norm": 0.20032985508441925,
2433
+ "learning_rate": 0.0008237609630751433,
2434
+ "loss": 2.8321,
2435
+ "step": 31100
2436
+ },
2437
+ {
2438
+ "epoch": 0.4051948051948052,
2439
+ "grad_norm": 0.2228243052959442,
2440
+ "learning_rate": 0.0008226609308270862,
2441
+ "loss": 2.8323,
2442
+ "step": 31200
2443
+ },
2444
+ {
2445
+ "epoch": 0.4064935064935065,
2446
+ "grad_norm": 0.22431324422359467,
2447
+ "learning_rate": 0.0008215582157601267,
2448
+ "loss": 2.8489,
2449
+ "step": 31300
2450
+ },
2451
+ {
2452
+ "epoch": 0.4077922077922078,
2453
+ "grad_norm": 0.20516513288021088,
2454
+ "learning_rate": 0.0008204528270429752,
2455
+ "loss": 2.8001,
2456
+ "step": 31400
2457
+ },
2458
+ {
2459
+ "epoch": 0.4090909090909091,
2460
+ "grad_norm": 0.20168229937553406,
2461
+ "learning_rate": 0.0008193447738665735,
2462
+ "loss": 2.8463,
2463
+ "step": 31500
2464
+ },
2465
+ {
2466
+ "epoch": 0.4103896103896104,
2467
+ "grad_norm": 0.20102174580097198,
2468
+ "learning_rate": 0.0008182340654440173,
2469
+ "loss": 2.8226,
2470
+ "step": 31600
2471
+ },
2472
+ {
2473
+ "epoch": 0.4116883116883117,
2474
+ "grad_norm": 0.2369510680437088,
2475
+ "learning_rate": 0.0008171207110104797,
2476
+ "loss": 2.8284,
2477
+ "step": 31700
2478
+ },
2479
+ {
2480
+ "epoch": 0.412987012987013,
2481
+ "grad_norm": 0.18998616933822632,
2482
+ "learning_rate": 0.0008160047198231344,
2483
+ "loss": 2.8506,
2484
+ "step": 31800
2485
+ },
2486
+ {
2487
+ "epoch": 0.4142857142857143,
2488
+ "grad_norm": 0.21406228840351105,
2489
+ "learning_rate": 0.000814886101161079,
2490
+ "loss": 2.81,
2491
+ "step": 31900
2492
+ },
2493
+ {
2494
+ "epoch": 0.4155844155844156,
2495
+ "grad_norm": 0.20903189480304718,
2496
+ "learning_rate": 0.0008137648643252575,
2497
+ "loss": 2.7976,
2498
+ "step": 32000
2499
+ },
2500
+ {
2501
+ "epoch": 0.4155844155844156,
2502
+ "eval_loss": 3.1785833835601807,
2503
+ "eval_runtime": 15.2674,
2504
+ "eval_samples_per_second": 37.727,
2505
+ "eval_steps_per_second": 9.432,
2506
+ "step": 32000
2507
  }
2508
  ],
2509
  "logging_steps": 100,
 
2523
  "attributes": {}
2524
  }
2525
  },
2526
+ "total_flos": 9.5261804199936e+17,
2527
  "train_batch_size": 22,
2528
  "trial_name": null,
2529
  "trial_params": null