CodeIsAbstract commited on
Commit
9997bb3
·
verified ·
1 Parent(s): 7ebc744

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4a67d81e97596f2768511e4959660a99c3dabff68129710f50333a180c97ccf4
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6474fca5a3c5baba3de6445b1ebdccd489d640243fc43690efa828d0c4fe3868
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:153578390500094706b5b2f9069514e099e3fa4d0f8671ad39f7456830573721
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e45d306cbed8140a90bd411a1fece5d9556ba1ea6e19977f26931c3c04004bd4
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:76cd281a3ce995fc434ac757f1094066d955dd6b7188cf148f0dd1523cb9cd80
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1b8777afaf65f04eb18ddaf41f68fd24275d3e886bf75d506bd86a6d10df7f8
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:350811cf0b8d7ebb0375f99d7c5be37c2b3fc48fc087f648711ef9871266ac16
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dfc5c6949cf8a9eca4d88710f0c9c06f410fb63454fb60c644727889ad457648
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.1,
6
  "eval_steps": 100,
7
- "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -398,6 +398,396 @@
398
  "eval_samples_per_second": 67.447,
399
  "eval_steps_per_second": 3.848,
400
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
401
  }
402
  ],
403
  "logging_steps": 10,
@@ -417,7 +807,7 @@
417
  "attributes": {}
418
  }
419
  },
420
- "total_flos": 1.63077257428992e+17,
421
  "train_batch_size": 18,
422
  "trial_name": null,
423
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2,
6
  "eval_steps": 100,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
398
  "eval_samples_per_second": 67.447,
399
  "eval_steps_per_second": 3.848,
400
  "step": 500
401
+ },
402
+ {
403
+ "epoch": 0.102,
404
+ "grad_norm": 0.66796875,
405
+ "learning_rate": 0.0003,
406
+ "loss": 2.6653621673583983,
407
+ "step": 510
408
+ },
409
+ {
410
+ "epoch": 0.104,
411
+ "grad_norm": 2.53125,
412
+ "learning_rate": 0.0003,
413
+ "loss": 2.6938413619995116,
414
+ "step": 520
415
+ },
416
+ {
417
+ "epoch": 0.106,
418
+ "grad_norm": 1.578125,
419
+ "learning_rate": 0.0003,
420
+ "loss": 2.6589494705200196,
421
+ "step": 530
422
+ },
423
+ {
424
+ "epoch": 0.108,
425
+ "grad_norm": 1.453125,
426
+ "learning_rate": 0.0003,
427
+ "loss": 2.6775186538696287,
428
+ "step": 540
429
+ },
430
+ {
431
+ "epoch": 0.11,
432
+ "grad_norm": 0.1806640625,
433
+ "learning_rate": 0.0003,
434
+ "loss": 2.659619140625,
435
+ "step": 550
436
+ },
437
+ {
438
+ "epoch": 0.112,
439
+ "grad_norm": 0.212890625,
440
+ "learning_rate": 0.0003,
441
+ "loss": 2.644103240966797,
442
+ "step": 560
443
+ },
444
+ {
445
+ "epoch": 0.114,
446
+ "grad_norm": 0.73046875,
447
+ "learning_rate": 0.0003,
448
+ "loss": 2.65213565826416,
449
+ "step": 570
450
+ },
451
+ {
452
+ "epoch": 0.116,
453
+ "grad_norm": 1.1796875,
454
+ "learning_rate": 0.0003,
455
+ "loss": 2.6230279922485353,
456
+ "step": 580
457
+ },
458
+ {
459
+ "epoch": 0.118,
460
+ "grad_norm": 0.166015625,
461
+ "learning_rate": 0.0003,
462
+ "loss": 2.6443111419677736,
463
+ "step": 590
464
+ },
465
+ {
466
+ "epoch": 0.12,
467
+ "grad_norm": 0.8203125,
468
+ "learning_rate": 0.0003,
469
+ "loss": 2.6569570541381835,
470
+ "step": 600
471
+ },
472
+ {
473
+ "epoch": 0.12,
474
+ "eval_loss": 3.0349245071411133,
475
+ "eval_runtime": 4.4206,
476
+ "eval_samples_per_second": 67.412,
477
+ "eval_steps_per_second": 3.846,
478
+ "step": 600
479
+ },
480
+ {
481
+ "epoch": 0.122,
482
+ "grad_norm": 0.1181640625,
483
+ "learning_rate": 0.0003,
484
+ "loss": 2.6595781326293944,
485
+ "step": 610
486
+ },
487
+ {
488
+ "epoch": 0.124,
489
+ "grad_norm": 0.232421875,
490
+ "learning_rate": 0.0003,
491
+ "loss": 2.610936737060547,
492
+ "step": 620
493
+ },
494
+ {
495
+ "epoch": 0.126,
496
+ "grad_norm": 0.1435546875,
497
+ "learning_rate": 0.0003,
498
+ "loss": 2.6387302398681642,
499
+ "step": 630
500
+ },
501
+ {
502
+ "epoch": 0.128,
503
+ "grad_norm": 0.14453125,
504
+ "learning_rate": 0.0003,
505
+ "loss": 2.62979793548584,
506
+ "step": 640
507
+ },
508
+ {
509
+ "epoch": 0.13,
510
+ "grad_norm": 2.34375,
511
+ "learning_rate": 0.0003,
512
+ "loss": 2.649636650085449,
513
+ "step": 650
514
+ },
515
+ {
516
+ "epoch": 0.132,
517
+ "grad_norm": 0.2041015625,
518
+ "learning_rate": 0.0003,
519
+ "loss": 2.622520637512207,
520
+ "step": 660
521
+ },
522
+ {
523
+ "epoch": 0.134,
524
+ "grad_norm": 0.97265625,
525
+ "learning_rate": 0.0003,
526
+ "loss": 2.646112823486328,
527
+ "step": 670
528
+ },
529
+ {
530
+ "epoch": 0.136,
531
+ "grad_norm": 5.8125,
532
+ "learning_rate": 0.0003,
533
+ "loss": 2.6434356689453127,
534
+ "step": 680
535
+ },
536
+ {
537
+ "epoch": 0.138,
538
+ "grad_norm": 0.10693359375,
539
+ "learning_rate": 0.0003,
540
+ "loss": 2.623065376281738,
541
+ "step": 690
542
+ },
543
+ {
544
+ "epoch": 0.14,
545
+ "grad_norm": 0.90625,
546
+ "learning_rate": 0.0003,
547
+ "loss": 2.6090648651123045,
548
+ "step": 700
549
+ },
550
+ {
551
+ "epoch": 0.14,
552
+ "eval_loss": 3.0306036472320557,
553
+ "eval_runtime": 4.4176,
554
+ "eval_samples_per_second": 67.457,
555
+ "eval_steps_per_second": 3.848,
556
+ "step": 700
557
+ },
558
+ {
559
+ "epoch": 0.142,
560
+ "grad_norm": 0.6328125,
561
+ "learning_rate": 0.0003,
562
+ "loss": 2.6384559631347657,
563
+ "step": 710
564
+ },
565
+ {
566
+ "epoch": 0.144,
567
+ "grad_norm": 0.640625,
568
+ "learning_rate": 0.0003,
569
+ "loss": 2.6472700119018553,
570
+ "step": 720
571
+ },
572
+ {
573
+ "epoch": 0.146,
574
+ "grad_norm": 0.07177734375,
575
+ "learning_rate": 0.0003,
576
+ "loss": 2.6483619689941404,
577
+ "step": 730
578
+ },
579
+ {
580
+ "epoch": 0.148,
581
+ "grad_norm": 0.322265625,
582
+ "learning_rate": 0.0003,
583
+ "loss": 2.626618766784668,
584
+ "step": 740
585
+ },
586
+ {
587
+ "epoch": 0.15,
588
+ "grad_norm": 0.099609375,
589
+ "learning_rate": 0.0003,
590
+ "loss": 2.6325525283813476,
591
+ "step": 750
592
+ },
593
+ {
594
+ "epoch": 0.152,
595
+ "grad_norm": 0.1611328125,
596
+ "learning_rate": 0.0003,
597
+ "loss": 2.6193023681640626,
598
+ "step": 760
599
+ },
600
+ {
601
+ "epoch": 0.154,
602
+ "grad_norm": 0.072265625,
603
+ "learning_rate": 0.0003,
604
+ "loss": 2.618332862854004,
605
+ "step": 770
606
+ },
607
+ {
608
+ "epoch": 0.156,
609
+ "grad_norm": 0.87109375,
610
+ "learning_rate": 0.0003,
611
+ "loss": 2.610393524169922,
612
+ "step": 780
613
+ },
614
+ {
615
+ "epoch": 0.158,
616
+ "grad_norm": 0.09716796875,
617
+ "learning_rate": 0.0003,
618
+ "loss": 2.6242008209228516,
619
+ "step": 790
620
+ },
621
+ {
622
+ "epoch": 0.16,
623
+ "grad_norm": 0.072265625,
624
+ "learning_rate": 0.0003,
625
+ "loss": 2.638929748535156,
626
+ "step": 800
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "eval_loss": 3.0286974906921387,
631
+ "eval_runtime": 4.4633,
632
+ "eval_samples_per_second": 66.766,
633
+ "eval_steps_per_second": 3.809,
634
+ "step": 800
635
+ },
636
+ {
637
+ "epoch": 0.162,
638
+ "grad_norm": 0.29296875,
639
+ "learning_rate": 0.0003,
640
+ "loss": 2.638911247253418,
641
+ "step": 810
642
+ },
643
+ {
644
+ "epoch": 0.164,
645
+ "grad_norm": 0.62890625,
646
+ "learning_rate": 0.0003,
647
+ "loss": 2.6282770156860353,
648
+ "step": 820
649
+ },
650
+ {
651
+ "epoch": 0.166,
652
+ "grad_norm": 0.0859375,
653
+ "learning_rate": 0.0003,
654
+ "loss": 2.668540382385254,
655
+ "step": 830
656
+ },
657
+ {
658
+ "epoch": 0.168,
659
+ "grad_norm": 0.08349609375,
660
+ "learning_rate": 0.0003,
661
+ "loss": 2.6025644302368165,
662
+ "step": 840
663
+ },
664
+ {
665
+ "epoch": 0.17,
666
+ "grad_norm": 0.251953125,
667
+ "learning_rate": 0.0003,
668
+ "loss": 2.614345932006836,
669
+ "step": 850
670
+ },
671
+ {
672
+ "epoch": 0.172,
673
+ "grad_norm": 1.2421875,
674
+ "learning_rate": 0.0003,
675
+ "loss": 2.6097484588623048,
676
+ "step": 860
677
+ },
678
+ {
679
+ "epoch": 0.174,
680
+ "grad_norm": 0.10107421875,
681
+ "learning_rate": 0.0003,
682
+ "loss": 2.5976139068603517,
683
+ "step": 870
684
+ },
685
+ {
686
+ "epoch": 0.176,
687
+ "grad_norm": 0.0634765625,
688
+ "learning_rate": 0.0003,
689
+ "loss": 2.609291648864746,
690
+ "step": 880
691
+ },
692
+ {
693
+ "epoch": 0.178,
694
+ "grad_norm": 0.060302734375,
695
+ "learning_rate": 0.0003,
696
+ "loss": 2.6136356353759767,
697
+ "step": 890
698
+ },
699
+ {
700
+ "epoch": 0.18,
701
+ "grad_norm": 0.306640625,
702
+ "learning_rate": 0.0003,
703
+ "loss": 2.6354192733764648,
704
+ "step": 900
705
+ },
706
+ {
707
+ "epoch": 0.18,
708
+ "eval_loss": 3.028738021850586,
709
+ "eval_runtime": 4.4125,
710
+ "eval_samples_per_second": 67.536,
711
+ "eval_steps_per_second": 3.853,
712
+ "step": 900
713
+ },
714
+ {
715
+ "epoch": 0.182,
716
+ "grad_norm": 0.65234375,
717
+ "learning_rate": 0.0003,
718
+ "loss": 2.6325847625732424,
719
+ "step": 910
720
+ },
721
+ {
722
+ "epoch": 0.184,
723
+ "grad_norm": 0.4921875,
724
+ "learning_rate": 0.0003,
725
+ "loss": 2.6379472732543947,
726
+ "step": 920
727
+ },
728
+ {
729
+ "epoch": 0.186,
730
+ "grad_norm": 0.216796875,
731
+ "learning_rate": 0.0003,
732
+ "loss": 2.647774505615234,
733
+ "step": 930
734
+ },
735
+ {
736
+ "epoch": 0.188,
737
+ "grad_norm": 0.083984375,
738
+ "learning_rate": 0.0003,
739
+ "loss": 2.628580665588379,
740
+ "step": 940
741
+ },
742
+ {
743
+ "epoch": 0.19,
744
+ "grad_norm": 0.09033203125,
745
+ "learning_rate": 0.0003,
746
+ "loss": 2.63638973236084,
747
+ "step": 950
748
+ },
749
+ {
750
+ "epoch": 0.192,
751
+ "grad_norm": 0.08056640625,
752
+ "learning_rate": 0.0003,
753
+ "loss": 2.628683853149414,
754
+ "step": 960
755
+ },
756
+ {
757
+ "epoch": 0.194,
758
+ "grad_norm": 0.06787109375,
759
+ "learning_rate": 0.0003,
760
+ "loss": 2.628560256958008,
761
+ "step": 970
762
+ },
763
+ {
764
+ "epoch": 0.196,
765
+ "grad_norm": 0.1328125,
766
+ "learning_rate": 0.0003,
767
+ "loss": 2.627498435974121,
768
+ "step": 980
769
+ },
770
+ {
771
+ "epoch": 0.198,
772
+ "grad_norm": 0.1796875,
773
+ "learning_rate": 0.0003,
774
+ "loss": 2.613533782958984,
775
+ "step": 990
776
+ },
777
+ {
778
+ "epoch": 0.2,
779
+ "grad_norm": 0.265625,
780
+ "learning_rate": 0.0003,
781
+ "loss": 2.615587615966797,
782
+ "step": 1000
783
+ },
784
+ {
785
+ "epoch": 0.2,
786
+ "eval_loss": 3.0314836502075195,
787
+ "eval_runtime": 4.4355,
788
+ "eval_samples_per_second": 67.186,
789
+ "eval_steps_per_second": 3.833,
790
+ "step": 1000
791
  }
792
  ],
793
  "logging_steps": 10,
 
807
  "attributes": {}
808
  }
809
  },
810
+ "total_flos": 3.26154514857984e+17,
811
  "train_batch_size": 18,
812
  "trial_name": null,
813
  "trial_params": null