CodeIsAbstract commited on
Commit
fda118e
·
verified ·
1 Parent(s): d6328d9

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:195264f4189d85c3b420f3c7ede63faa40b60d80bcb0a67ef9a76fbfa01db140
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58d40820fb221ed5ab5ca4d9701da2049c6d1761cfc543431f711239f4c0a14f
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0f6addff14bf805b5812fe850aec8f4687d039d8b226f53eba8e2bd87eb56179
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5169ba6435acf0ba6f9c584893daa0bc9782c7d8adc4d46b177c96492031e90b
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:718a0f3db00824213036a2c0441849791319b7d9cf189065873bb26a7020738e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:363c5df1543d2c82b2f13164f35bdd0367ceb32e7fa1b2f67c19df073a08b17b
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:636ce205a7d25245e8bf516169992a8697e5d28e02c0cbb97c0ae7b282b76a7d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4abbae167a9f0ed409523b108568bee9d935380e25b792884224d262a0c2c795
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.05194805194805195,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -320,6 +320,318 @@
320
  "eval_samples_per_second": 34.081,
321
  "eval_steps_per_second": 8.52,
322
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
323
  }
324
  ],
325
  "logging_steps": 100,
@@ -339,7 +651,7 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 1.1907725524992e+17,
343
  "train_batch_size": 22,
344
  "trial_name": null,
345
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.1038961038961039,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
320
  "eval_samples_per_second": 34.081,
321
  "eval_steps_per_second": 8.52,
322
  "step": 4000
323
+ },
324
+ {
325
+ "epoch": 0.053246753246753244,
326
+ "grad_norm": 0.24371765553951263,
327
+ "learning_rate": 0.0009980688306054227,
328
+ "loss": 3.2775,
329
+ "step": 4100
330
+ },
331
+ {
332
+ "epoch": 0.05454545454545454,
333
+ "grad_norm": 0.20401820540428162,
334
+ "learning_rate": 0.0009979401661574985,
335
+ "loss": 3.2952,
336
+ "step": 4200
337
+ },
338
+ {
339
+ "epoch": 0.05584415584415584,
340
+ "grad_norm": 0.2212119996547699,
341
+ "learning_rate": 0.0009978073615018127,
342
+ "loss": 3.257,
343
+ "step": 4300
344
+ },
345
+ {
346
+ "epoch": 0.05714285714285714,
347
+ "grad_norm": 0.2384602129459381,
348
+ "learning_rate": 0.000997670417742592,
349
+ "loss": 3.2859,
350
+ "step": 4400
351
+ },
352
+ {
353
+ "epoch": 0.05844155844155844,
354
+ "grad_norm": 0.21143975853919983,
355
+ "learning_rate": 0.0009975293360184785,
356
+ "loss": 3.2475,
357
+ "step": 4500
358
+ },
359
+ {
360
+ "epoch": 0.05974025974025974,
361
+ "grad_norm": 0.20411203801631927,
362
+ "learning_rate": 0.00099738411750252,
363
+ "loss": 3.2443,
364
+ "step": 4600
365
+ },
366
+ {
367
+ "epoch": 0.06103896103896104,
368
+ "grad_norm": 0.2097695916891098,
369
+ "learning_rate": 0.0009972347634021606,
370
+ "loss": 3.2703,
371
+ "step": 4700
372
+ },
373
+ {
374
+ "epoch": 0.06233766233766234,
375
+ "grad_norm": 0.2140115350484848,
376
+ "learning_rate": 0.00099708127495923,
377
+ "loss": 3.2569,
378
+ "step": 4800
379
+ },
380
+ {
381
+ "epoch": 0.06363636363636363,
382
+ "grad_norm": 0.2163541465997696,
383
+ "learning_rate": 0.000996923653449934,
384
+ "loss": 3.2646,
385
+ "step": 4900
386
+ },
387
+ {
388
+ "epoch": 0.06493506493506493,
389
+ "grad_norm": 0.1931372582912445,
390
+ "learning_rate": 0.0009967619001848434,
391
+ "loss": 3.2224,
392
+ "step": 5000
393
+ },
394
+ {
395
+ "epoch": 0.06493506493506493,
396
+ "eval_loss": 3.59517240524292,
397
+ "eval_runtime": 15.382,
398
+ "eval_samples_per_second": 37.446,
399
+ "eval_steps_per_second": 9.362,
400
+ "step": 5000
401
+ },
402
+ {
403
+ "epoch": 0.06623376623376623,
404
+ "grad_norm": 0.2017965465784073,
405
+ "learning_rate": 0.0009965960165088828,
406
+ "loss": 3.2159,
407
+ "step": 5100
408
+ },
409
+ {
410
+ "epoch": 0.06753246753246753,
411
+ "grad_norm": 0.2096720039844513,
412
+ "learning_rate": 0.0009964260038013203,
413
+ "loss": 3.1967,
414
+ "step": 5200
415
+ },
416
+ {
417
+ "epoch": 0.06883116883116883,
418
+ "grad_norm": 0.2119290977716446,
419
+ "learning_rate": 0.000996251863475755,
420
+ "loss": 3.1962,
421
+ "step": 5300
422
+ },
423
+ {
424
+ "epoch": 0.07012987012987013,
425
+ "grad_norm": 0.27017727494239807,
426
+ "learning_rate": 0.0009960735969801067,
427
+ "loss": 3.1672,
428
+ "step": 5400
429
+ },
430
+ {
431
+ "epoch": 0.07142857142857142,
432
+ "grad_norm": 0.21126972138881683,
433
+ "learning_rate": 0.0009958912057966018,
434
+ "loss": 3.2301,
435
+ "step": 5500
436
+ },
437
+ {
438
+ "epoch": 0.07272727272727272,
439
+ "grad_norm": 0.238760307431221,
440
+ "learning_rate": 0.0009957046914417628,
441
+ "loss": 3.2367,
442
+ "step": 5600
443
+ },
444
+ {
445
+ "epoch": 0.07402597402597402,
446
+ "grad_norm": 0.19038313627243042,
447
+ "learning_rate": 0.000995514055466395,
448
+ "loss": 3.2026,
449
+ "step": 5700
450
+ },
451
+ {
452
+ "epoch": 0.07532467532467532,
453
+ "grad_norm": 0.20789292454719543,
454
+ "learning_rate": 0.0009953192994555736,
455
+ "loss": 3.1718,
456
+ "step": 5800
457
+ },
458
+ {
459
+ "epoch": 0.07662337662337662,
460
+ "grad_norm": 0.18676762282848358,
461
+ "learning_rate": 0.00099512042502863,
462
+ "loss": 3.167,
463
+ "step": 5900
464
+ },
465
+ {
466
+ "epoch": 0.07792207792207792,
467
+ "grad_norm": 0.19430765509605408,
468
+ "learning_rate": 0.0009949174338391394,
469
+ "loss": 3.1721,
470
+ "step": 6000
471
+ },
472
+ {
473
+ "epoch": 0.07792207792207792,
474
+ "eval_loss": 3.5258212089538574,
475
+ "eval_runtime": 15.3635,
476
+ "eval_samples_per_second": 37.491,
477
+ "eval_steps_per_second": 9.373,
478
+ "step": 6000
479
+ },
480
+ {
481
+ "epoch": 0.07922077922077922,
482
+ "grad_norm": 0.20157617330551147,
483
+ "learning_rate": 0.0009947103275749064,
484
+ "loss": 3.1696,
485
+ "step": 6100
486
+ },
487
+ {
488
+ "epoch": 0.08051948051948052,
489
+ "grad_norm": 0.18047969043254852,
490
+ "learning_rate": 0.0009944991079579514,
491
+ "loss": 3.1601,
492
+ "step": 6200
493
+ },
494
+ {
495
+ "epoch": 0.08181818181818182,
496
+ "grad_norm": 0.2030571699142456,
497
+ "learning_rate": 0.0009942837767444952,
498
+ "loss": 3.1709,
499
+ "step": 6300
500
+ },
501
+ {
502
+ "epoch": 0.08311688311688312,
503
+ "grad_norm": 0.18791615962982178,
504
+ "learning_rate": 0.000994064335724946,
505
+ "loss": 3.1808,
506
+ "step": 6400
507
+ },
508
+ {
509
+ "epoch": 0.08441558441558442,
510
+ "grad_norm": 0.1927482783794403,
511
+ "learning_rate": 0.0009938407867238828,
512
+ "loss": 3.1272,
513
+ "step": 6500
514
+ },
515
+ {
516
+ "epoch": 0.08571428571428572,
517
+ "grad_norm": 0.1764034628868103,
518
+ "learning_rate": 0.0009936131316000418,
519
+ "loss": 3.1407,
520
+ "step": 6600
521
+ },
522
+ {
523
+ "epoch": 0.08701298701298701,
524
+ "grad_norm": 0.1786874383687973,
525
+ "learning_rate": 0.0009933813722463004,
526
+ "loss": 3.107,
527
+ "step": 6700
528
+ },
529
+ {
530
+ "epoch": 0.08831168831168831,
531
+ "grad_norm": 0.2685558497905731,
532
+ "learning_rate": 0.0009931455105896606,
533
+ "loss": 3.1609,
534
+ "step": 6800
535
+ },
536
+ {
537
+ "epoch": 0.08961038961038961,
538
+ "grad_norm": 0.19291436672210693,
539
+ "learning_rate": 0.000992905548591234,
540
+ "loss": 3.1444,
541
+ "step": 6900
542
+ },
543
+ {
544
+ "epoch": 0.09090909090909091,
545
+ "grad_norm": 0.1896260678768158,
546
+ "learning_rate": 0.0009926614882462253,
547
+ "loss": 3.1329,
548
+ "step": 7000
549
+ },
550
+ {
551
+ "epoch": 0.09090909090909091,
552
+ "eval_loss": 3.4966745376586914,
553
+ "eval_runtime": 14.4834,
554
+ "eval_samples_per_second": 39.77,
555
+ "eval_steps_per_second": 9.942,
556
+ "step": 7000
557
+ },
558
+ {
559
+ "epoch": 0.09220779220779221,
560
+ "grad_norm": 0.17750093340873718,
561
+ "learning_rate": 0.0009924133315839156,
562
+ "loss": 3.1602,
563
+ "step": 7100
564
+ },
565
+ {
566
+ "epoch": 0.09350649350649351,
567
+ "grad_norm": 0.18574649095535278,
568
+ "learning_rate": 0.0009921610806676456,
569
+ "loss": 3.1451,
570
+ "step": 7200
571
+ },
572
+ {
573
+ "epoch": 0.09480519480519481,
574
+ "grad_norm": 0.18256109952926636,
575
+ "learning_rate": 0.000991904737594798,
576
+ "loss": 3.1305,
577
+ "step": 7300
578
+ },
579
+ {
580
+ "epoch": 0.09610389610389611,
581
+ "grad_norm": 0.17488889396190643,
582
+ "learning_rate": 0.0009916443044967807,
583
+ "loss": 3.1211,
584
+ "step": 7400
585
+ },
586
+ {
587
+ "epoch": 0.09740259740259741,
588
+ "grad_norm": 0.205893412232399,
589
+ "learning_rate": 0.0009913797835390088,
590
+ "loss": 3.1039,
591
+ "step": 7500
592
+ },
593
+ {
594
+ "epoch": 0.0987012987012987,
595
+ "grad_norm": 0.20423896610736847,
596
+ "learning_rate": 0.0009911111769208864,
597
+ "loss": 3.1159,
598
+ "step": 7600
599
+ },
600
+ {
601
+ "epoch": 0.1,
602
+ "grad_norm": 0.18055465817451477,
603
+ "learning_rate": 0.000990838486875789,
604
+ "loss": 3.0872,
605
+ "step": 7700
606
+ },
607
+ {
608
+ "epoch": 0.1012987012987013,
609
+ "grad_norm": 0.18993014097213745,
610
+ "learning_rate": 0.0009905617156710437,
611
+ "loss": 3.0976,
612
+ "step": 7800
613
+ },
614
+ {
615
+ "epoch": 0.1025974025974026,
616
+ "grad_norm": 0.1896851807832718,
617
+ "learning_rate": 0.000990280865607912,
618
+ "loss": 3.0932,
619
+ "step": 7900
620
+ },
621
+ {
622
+ "epoch": 0.1038961038961039,
623
+ "grad_norm": 0.18168048560619354,
624
+ "learning_rate": 0.0009899959390215689,
625
+ "loss": 3.0862,
626
+ "step": 8000
627
+ },
628
+ {
629
+ "epoch": 0.1038961038961039,
630
+ "eval_loss": 3.4424216747283936,
631
+ "eval_runtime": 12.6555,
632
+ "eval_samples_per_second": 45.514,
633
+ "eval_steps_per_second": 11.378,
634
+ "step": 8000
635
  }
636
  ],
637
  "logging_steps": 100,
 
651
  "attributes": {}
652
  }
653
  },
654
+ "total_flos": 2.3815451049984e+17,
655
  "train_batch_size": 22,
656
  "trial_name": null,
657
  "trial_params": null