CodeIsAbstract commited on
Commit
564359d
·
verified ·
1 Parent(s): e0aae1e

Training in progress, step 36000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3fea0a0b88a8852aec2382a6400435ae84bbfcb52cd14372fe08dab6f5a9b181
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e271f094dc095c867358bb6157221396d158922b11c316900c8d79c1e09d0f6
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5bbb61cfa227633b44a69a3c624c1b0dbf80db3ba79898d7b764dc31de1163ee
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:55e5fe87fe81d7cb83f3ed0295308626dfebead62fc3b096c5571405aeeab756
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:428aed83766f09a6a2d9ccfb86972eeda646225b3b0cd6accec4a90e4f900161
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87b07b911601d04abcbcc91c12340e18730f0dbec6517e92e9000c51517972c9
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:35deceb007dc05c0a67e4f7d57b32921e37176fa5a3cb5bba0983b1b73cb3efc
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9eea36234165600a5923f1ab0a2e7e9de4f1d318a06ca4c532030966bdcc676
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4155844155844156,
6
  "eval_steps": 1000,
7
- "global_step": 32000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2504,6 +2504,318 @@
2504
  "eval_samples_per_second": 37.727,
2505
  "eval_steps_per_second": 9.432,
2506
  "step": 32000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2507
  }
2508
  ],
2509
  "logging_steps": 100,
@@ -2523,7 +2835,7 @@
2523
  "attributes": {}
2524
  }
2525
  },
2526
- "total_flos": 9.5261804199936e+17,
2527
  "train_batch_size": 22,
2528
  "trial_name": null,
2529
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4675324675324675,
6
  "eval_steps": 1000,
7
+ "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2504
  "eval_samples_per_second": 37.727,
2505
  "eval_steps_per_second": 9.432,
2506
  "step": 32000
2507
+ },
2508
+ {
2509
+ "epoch": 0.41688311688311686,
2510
+ "grad_norm": 0.22132746875286102,
2511
+ "learning_rate": 0.0008126410186383836,
2512
+ "loss": 2.8195,
2513
+ "step": 32100
2514
+ },
2515
+ {
2516
+ "epoch": 0.41818181818181815,
2517
+ "grad_norm": 0.21235358715057373,
2518
+ "learning_rate": 0.0008115145734448622,
2519
+ "loss": 2.7824,
2520
+ "step": 32200
2521
+ },
2522
+ {
2523
+ "epoch": 0.41948051948051945,
2524
+ "grad_norm": 0.21031178534030914,
2525
+ "learning_rate": 0.0008103855381107122,
2526
+ "loss": 2.8135,
2527
+ "step": 32300
2528
+ },
2529
+ {
2530
+ "epoch": 0.42077922077922075,
2531
+ "grad_norm": 0.2186926305294037,
2532
+ "learning_rate": 0.0008092539220234894,
2533
+ "loss": 2.8166,
2534
+ "step": 32400
2535
+ },
2536
+ {
2537
+ "epoch": 0.42207792207792205,
2538
+ "grad_norm": 0.20855864882469177,
2539
+ "learning_rate": 0.0008081197345922069,
2540
+ "loss": 2.8595,
2541
+ "step": 32500
2542
+ },
2543
+ {
2544
+ "epoch": 0.42337662337662335,
2545
+ "grad_norm": 0.20601770281791687,
2546
+ "learning_rate": 0.0008069829852472583,
2547
+ "loss": 2.8201,
2548
+ "step": 32600
2549
+ },
2550
+ {
2551
+ "epoch": 0.42467532467532465,
2552
+ "grad_norm": 0.22806823253631592,
2553
+ "learning_rate": 0.000805843683440338,
2554
+ "loss": 2.8265,
2555
+ "step": 32700
2556
+ },
2557
+ {
2558
+ "epoch": 0.42597402597402595,
2559
+ "grad_norm": 0.21018485724925995,
2560
+ "learning_rate": 0.0008047018386443638,
2561
+ "loss": 2.8403,
2562
+ "step": 32800
2563
+ },
2564
+ {
2565
+ "epoch": 0.42727272727272725,
2566
+ "grad_norm": 0.218631774187088,
2567
+ "learning_rate": 0.0008035574603533975,
2568
+ "loss": 2.8052,
2569
+ "step": 32900
2570
+ },
2571
+ {
2572
+ "epoch": 0.42857142857142855,
2573
+ "grad_norm": 0.20622466504573822,
2574
+ "learning_rate": 0.000802410558082566,
2575
+ "loss": 2.824,
2576
+ "step": 33000
2577
+ },
2578
+ {
2579
+ "epoch": 0.42857142857142855,
2580
+ "eval_loss": 3.178421974182129,
2581
+ "eval_runtime": 15.6887,
2582
+ "eval_samples_per_second": 36.714,
2583
+ "eval_steps_per_second": 9.179,
2584
+ "step": 33000
2585
+ },
2586
+ {
2587
+ "epoch": 0.42987012987012985,
2588
+ "grad_norm": 0.24304358661174774,
2589
+ "learning_rate": 0.0008012611413679824,
2590
+ "loss": 2.8068,
2591
+ "step": 33100
2592
+ },
2593
+ {
2594
+ "epoch": 0.43116883116883115,
2595
+ "grad_norm": 0.20239757001399994,
2596
+ "learning_rate": 0.0008001092197666661,
2597
+ "loss": 2.7936,
2598
+ "step": 33200
2599
+ },
2600
+ {
2601
+ "epoch": 0.43246753246753245,
2602
+ "grad_norm": 0.21365991234779358,
2603
+ "learning_rate": 0.0007989548028564646,
2604
+ "loss": 2.8053,
2605
+ "step": 33300
2606
+ },
2607
+ {
2608
+ "epoch": 0.43376623376623374,
2609
+ "grad_norm": 0.19812121987342834,
2610
+ "learning_rate": 0.0007977979002359723,
2611
+ "loss": 2.8083,
2612
+ "step": 33400
2613
+ },
2614
+ {
2615
+ "epoch": 0.43506493506493504,
2616
+ "grad_norm": 0.20705953240394592,
2617
+ "learning_rate": 0.0007966385215244518,
2618
+ "loss": 2.7882,
2619
+ "step": 33500
2620
+ },
2621
+ {
2622
+ "epoch": 0.43636363636363634,
2623
+ "grad_norm": 0.23277799785137177,
2624
+ "learning_rate": 0.0007954766763617538,
2625
+ "loss": 2.8241,
2626
+ "step": 33600
2627
+ },
2628
+ {
2629
+ "epoch": 0.43766233766233764,
2630
+ "grad_norm": 0.2175544649362564,
2631
+ "learning_rate": 0.0007943123744082363,
2632
+ "loss": 2.816,
2633
+ "step": 33700
2634
+ },
2635
+ {
2636
+ "epoch": 0.43896103896103894,
2637
+ "grad_norm": 0.21744564175605774,
2638
+ "learning_rate": 0.000793145625344685,
2639
+ "loss": 2.7947,
2640
+ "step": 33800
2641
+ },
2642
+ {
2643
+ "epoch": 0.44025974025974024,
2644
+ "grad_norm": 0.2085115909576416,
2645
+ "learning_rate": 0.0007919764388722322,
2646
+ "loss": 2.8016,
2647
+ "step": 33900
2648
+ },
2649
+ {
2650
+ "epoch": 0.44155844155844154,
2651
+ "grad_norm": 0.21085582673549652,
2652
+ "learning_rate": 0.0007908048247122768,
2653
+ "loss": 2.7845,
2654
+ "step": 34000
2655
+ },
2656
+ {
2657
+ "epoch": 0.44155844155844154,
2658
+ "eval_loss": 3.1732594966888428,
2659
+ "eval_runtime": 14.9965,
2660
+ "eval_samples_per_second": 38.409,
2661
+ "eval_steps_per_second": 9.602,
2662
+ "step": 34000
2663
+ },
2664
+ {
2665
+ "epoch": 0.44285714285714284,
2666
+ "grad_norm": 0.1986636519432068,
2667
+ "learning_rate": 0.0007896307926064029,
2668
+ "loss": 2.8039,
2669
+ "step": 34100
2670
+ },
2671
+ {
2672
+ "epoch": 0.44415584415584414,
2673
+ "grad_norm": 0.2274903804063797,
2674
+ "learning_rate": 0.0007884543523162991,
2675
+ "loss": 2.7982,
2676
+ "step": 34200
2677
+ },
2678
+ {
2679
+ "epoch": 0.44545454545454544,
2680
+ "grad_norm": 0.23180291056632996,
2681
+ "learning_rate": 0.0007872755136236774,
2682
+ "loss": 2.8223,
2683
+ "step": 34300
2684
+ },
2685
+ {
2686
+ "epoch": 0.44675324675324674,
2687
+ "grad_norm": 0.21269173920154572,
2688
+ "learning_rate": 0.0007860942863301914,
2689
+ "loss": 2.8141,
2690
+ "step": 34400
2691
+ },
2692
+ {
2693
+ "epoch": 0.44805194805194803,
2694
+ "grad_norm": 0.20867326855659485,
2695
+ "learning_rate": 0.0007849106802573553,
2696
+ "loss": 2.7805,
2697
+ "step": 34500
2698
+ },
2699
+ {
2700
+ "epoch": 0.44935064935064933,
2701
+ "grad_norm": 1.6756339073181152,
2702
+ "learning_rate": 0.0007837247052464621,
2703
+ "loss": 2.7879,
2704
+ "step": 34600
2705
+ },
2706
+ {
2707
+ "epoch": 0.45064935064935063,
2708
+ "grad_norm": 0.24874623119831085,
2709
+ "learning_rate": 0.0007825363711585016,
2710
+ "loss": 2.8281,
2711
+ "step": 34700
2712
+ },
2713
+ {
2714
+ "epoch": 0.45194805194805193,
2715
+ "grad_norm": 0.19419661164283752,
2716
+ "learning_rate": 0.0007813456878740789,
2717
+ "loss": 2.7984,
2718
+ "step": 34800
2719
+ },
2720
+ {
2721
+ "epoch": 0.45324675324675323,
2722
+ "grad_norm": 0.21280324459075928,
2723
+ "learning_rate": 0.0007801526652933313,
2724
+ "loss": 2.834,
2725
+ "step": 34900
2726
+ },
2727
+ {
2728
+ "epoch": 0.45454545454545453,
2729
+ "grad_norm": 0.20888017117977142,
2730
+ "learning_rate": 0.0007789573133358471,
2731
+ "loss": 2.831,
2732
+ "step": 35000
2733
+ },
2734
+ {
2735
+ "epoch": 0.45454545454545453,
2736
+ "eval_loss": 3.1828813552856445,
2737
+ "eval_runtime": 14.929,
2738
+ "eval_samples_per_second": 38.583,
2739
+ "eval_steps_per_second": 9.646,
2740
+ "step": 35000
2741
+ },
2742
+ {
2743
+ "epoch": 0.45584415584415583,
2744
+ "grad_norm": 0.21340541541576385,
2745
+ "learning_rate": 0.0007777596419405823,
2746
+ "loss": 2.8065,
2747
+ "step": 35100
2748
+ },
2749
+ {
2750
+ "epoch": 0.45714285714285713,
2751
+ "grad_norm": 0.20886409282684326,
2752
+ "learning_rate": 0.0007765596610657783,
2753
+ "loss": 2.8423,
2754
+ "step": 35200
2755
+ },
2756
+ {
2757
+ "epoch": 0.4584415584415584,
2758
+ "grad_norm": 0.21619443595409393,
2759
+ "learning_rate": 0.0007753573806888795,
2760
+ "loss": 2.786,
2761
+ "step": 35300
2762
+ },
2763
+ {
2764
+ "epoch": 0.4597402597402597,
2765
+ "grad_norm": 0.21694345772266388,
2766
+ "learning_rate": 0.0007741528108064491,
2767
+ "loss": 2.7822,
2768
+ "step": 35400
2769
+ },
2770
+ {
2771
+ "epoch": 0.461038961038961,
2772
+ "grad_norm": 0.21181565523147583,
2773
+ "learning_rate": 0.0007729459614340872,
2774
+ "loss": 2.7599,
2775
+ "step": 35500
2776
+ },
2777
+ {
2778
+ "epoch": 0.4623376623376623,
2779
+ "grad_norm": 0.22892408072948456,
2780
+ "learning_rate": 0.0007717368426063475,
2781
+ "loss": 2.7934,
2782
+ "step": 35600
2783
+ },
2784
+ {
2785
+ "epoch": 0.4636363636363636,
2786
+ "grad_norm": 0.21864919364452362,
2787
+ "learning_rate": 0.0007705254643766527,
2788
+ "loss": 2.8059,
2789
+ "step": 35700
2790
+ },
2791
+ {
2792
+ "epoch": 0.4649350649350649,
2793
+ "grad_norm": 0.2145598828792572,
2794
+ "learning_rate": 0.000769311836817212,
2795
+ "loss": 2.8056,
2796
+ "step": 35800
2797
+ },
2798
+ {
2799
+ "epoch": 0.4662337662337662,
2800
+ "grad_norm": 0.21081195771694183,
2801
+ "learning_rate": 0.0007680959700189375,
2802
+ "loss": 2.7969,
2803
+ "step": 35900
2804
+ },
2805
+ {
2806
+ "epoch": 0.4675324675324675,
2807
+ "grad_norm": 0.22019855678081512,
2808
+ "learning_rate": 0.0007668778740913591,
2809
+ "loss": 2.7923,
2810
+ "step": 36000
2811
+ },
2812
+ {
2813
+ "epoch": 0.4675324675324675,
2814
+ "eval_loss": 3.161719560623169,
2815
+ "eval_runtime": 15.4038,
2816
+ "eval_samples_per_second": 37.393,
2817
+ "eval_steps_per_second": 9.348,
2818
+ "step": 36000
2819
  }
2820
  ],
2821
  "logging_steps": 100,
 
2835
  "attributes": {}
2836
  }
2837
  },
2838
+ "total_flos": 1.07169529724928e+18,
2839
  "train_batch_size": 22,
2840
  "trial_name": null,
2841
  "trial_params": null