CodeIsAbstract commited on
Commit
b0cc820
·
verified ·
1 Parent(s): 85b4649

Training in progress, step 24000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:424b403b05bd2545c2a14424fd014bcf57732428b67aad097d07f185a6bc5bb2
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac08b42599d1bde48be74806637c6c061e7123fe79086fdd5f3704e9fd3d6ce1
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b0098b151fd7135a628ba22af2fdd6e87cb9087d29b7db0fa633dfc8d1f2500a
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a587cba38b05004cfdcf48f4bea4b6d068091255cf02ec07324ae8e96318e4f0
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12d8b6c71ec5842a7f720763e6312f0db9384dc999ad47a74f64d26d1c1cb7ce
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a804dd9b4962bc1e7c8e5b51c83ce95f04ab0a366340b47fc4849e7d4ecffd6d
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bfa5d29ca2c7a5e4464c7ab7540a14b1b7182a94f039f78dc78638842bee89e8
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:668b318436d635cfc651258a221542355f21124096cff5d5c7e78a7d5c81ab33
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2597402597402597,
6
  "eval_steps": 1000,
7
- "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1568,6 +1568,318 @@
1568
  "eval_samples_per_second": 38.297,
1569
  "eval_steps_per_second": 9.574,
1570
  "step": 20000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1571
  }
1572
  ],
1573
  "logging_steps": 100,
@@ -1587,7 +1899,7 @@
1587
  "attributes": {}
1588
  }
1589
  },
1590
- "total_flos": 5.953862762496e+17,
1591
  "train_batch_size": 22,
1592
  "trial_name": null,
1593
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.3116883116883117,
6
  "eval_steps": 1000,
7
+ "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1568
  "eval_samples_per_second": 38.297,
1569
  "eval_steps_per_second": 9.574,
1570
  "step": 20000
1571
+ },
1572
+ {
1573
+ "epoch": 0.261038961038961,
1574
+ "grad_norm": 0.1809331625699997,
1575
+ "learning_rate": 0.0009264499766083365,
1576
+ "loss": 2.8836,
1577
+ "step": 20100
1578
+ },
1579
+ {
1580
+ "epoch": 0.2623376623376623,
1581
+ "grad_norm": 0.19160114228725433,
1582
+ "learning_rate": 0.0009256954993538877,
1583
+ "loss": 2.8881,
1584
+ "step": 20200
1585
+ },
1586
+ {
1587
+ "epoch": 0.2636363636363636,
1588
+ "grad_norm": 0.1818685233592987,
1589
+ "learning_rate": 0.0009249374825821832,
1590
+ "loss": 2.8543,
1591
+ "step": 20300
1592
+ },
1593
+ {
1594
+ "epoch": 0.2649350649350649,
1595
+ "grad_norm": 0.1778269112110138,
1596
+ "learning_rate": 0.0009241759325958815,
1597
+ "loss": 2.8741,
1598
+ "step": 20400
1599
+ },
1600
+ {
1601
+ "epoch": 0.2662337662337662,
1602
+ "grad_norm": 0.1888059824705124,
1603
+ "learning_rate": 0.0009234108557270187,
1604
+ "loss": 2.8772,
1605
+ "step": 20500
1606
+ },
1607
+ {
1608
+ "epoch": 0.2675324675324675,
1609
+ "grad_norm": 0.22666007280349731,
1610
+ "learning_rate": 0.000922642258336956,
1611
+ "loss": 2.9021,
1612
+ "step": 20600
1613
+ },
1614
+ {
1615
+ "epoch": 0.2688311688311688,
1616
+ "grad_norm": 0.18350407481193542,
1617
+ "learning_rate": 0.0009218701468163267,
1618
+ "loss": 2.8845,
1619
+ "step": 20700
1620
+ },
1621
+ {
1622
+ "epoch": 0.2701298701298701,
1623
+ "grad_norm": 0.17511753737926483,
1624
+ "learning_rate": 0.000921094527584982,
1625
+ "loss": 2.9005,
1626
+ "step": 20800
1627
+ },
1628
+ {
1629
+ "epoch": 0.2714285714285714,
1630
+ "grad_norm": 0.18808013200759888,
1631
+ "learning_rate": 0.0009203154070919398,
1632
+ "loss": 2.8658,
1633
+ "step": 20900
1634
+ },
1635
+ {
1636
+ "epoch": 0.2727272727272727,
1637
+ "grad_norm": 0.18752652406692505,
1638
+ "learning_rate": 0.0009195327918153292,
1639
+ "loss": 2.8649,
1640
+ "step": 21000
1641
+ },
1642
+ {
1643
+ "epoch": 0.2727272727272727,
1644
+ "eval_loss": 3.2498960494995117,
1645
+ "eval_runtime": 15.6914,
1646
+ "eval_samples_per_second": 36.708,
1647
+ "eval_steps_per_second": 9.177,
1648
+ "step": 21000
1649
+ },
1650
+ {
1651
+ "epoch": 0.274025974025974,
1652
+ "grad_norm": 0.19540126621723175,
1653
+ "learning_rate": 0.0009187466882623372,
1654
+ "loss": 2.8755,
1655
+ "step": 21100
1656
+ },
1657
+ {
1658
+ "epoch": 0.2753246753246753,
1659
+ "grad_norm": 0.18758288025856018,
1660
+ "learning_rate": 0.0009179571029691546,
1661
+ "loss": 2.8737,
1662
+ "step": 21200
1663
+ },
1664
+ {
1665
+ "epoch": 0.2766233766233766,
1666
+ "grad_norm": 0.18555940687656403,
1667
+ "learning_rate": 0.0009171640425009224,
1668
+ "loss": 2.8626,
1669
+ "step": 21300
1670
+ },
1671
+ {
1672
+ "epoch": 0.2779220779220779,
1673
+ "grad_norm": 0.18831279873847961,
1674
+ "learning_rate": 0.0009163675134516758,
1675
+ "loss": 2.883,
1676
+ "step": 21400
1677
+ },
1678
+ {
1679
+ "epoch": 0.2792207792207792,
1680
+ "grad_norm": 0.19656601548194885,
1681
+ "learning_rate": 0.0009155675224442904,
1682
+ "loss": 2.8959,
1683
+ "step": 21500
1684
+ },
1685
+ {
1686
+ "epoch": 0.2805194805194805,
1687
+ "grad_norm": 0.18778946995735168,
1688
+ "learning_rate": 0.0009147640761304266,
1689
+ "loss": 2.8723,
1690
+ "step": 21600
1691
+ },
1692
+ {
1693
+ "epoch": 0.2818181818181818,
1694
+ "grad_norm": 0.1975911557674408,
1695
+ "learning_rate": 0.000913957181190475,
1696
+ "loss": 2.8724,
1697
+ "step": 21700
1698
+ },
1699
+ {
1700
+ "epoch": 0.2831168831168831,
1701
+ "grad_norm": 0.18618452548980713,
1702
+ "learning_rate": 0.0009131468443334998,
1703
+ "loss": 2.8597,
1704
+ "step": 21800
1705
+ },
1706
+ {
1707
+ "epoch": 0.2844155844155844,
1708
+ "grad_norm": 0.19419926404953003,
1709
+ "learning_rate": 0.0009123330722971841,
1710
+ "loss": 2.8398,
1711
+ "step": 21900
1712
+ },
1713
+ {
1714
+ "epoch": 0.2857142857142857,
1715
+ "grad_norm": 0.1885763257741928,
1716
+ "learning_rate": 0.0009115158718477732,
1717
+ "loss": 2.875,
1718
+ "step": 22000
1719
+ },
1720
+ {
1721
+ "epoch": 0.2857142857142857,
1722
+ "eval_loss": 3.2455015182495117,
1723
+ "eval_runtime": 14.7497,
1724
+ "eval_samples_per_second": 39.052,
1725
+ "eval_steps_per_second": 9.763,
1726
+ "step": 22000
1727
+ },
1728
+ {
1729
+ "epoch": 0.287012987012987,
1730
+ "grad_norm": 0.17821335792541504,
1731
+ "learning_rate": 0.0009106952497800183,
1732
+ "loss": 2.8378,
1733
+ "step": 22100
1734
+ },
1735
+ {
1736
+ "epoch": 0.2883116883116883,
1737
+ "grad_norm": 0.20162396132946014,
1738
+ "learning_rate": 0.0009098712129171207,
1739
+ "loss": 2.8512,
1740
+ "step": 22200
1741
+ },
1742
+ {
1743
+ "epoch": 0.2896103896103896,
1744
+ "grad_norm": 0.21083980798721313,
1745
+ "learning_rate": 0.0009090437681106742,
1746
+ "loss": 2.8587,
1747
+ "step": 22300
1748
+ },
1749
+ {
1750
+ "epoch": 0.2909090909090909,
1751
+ "grad_norm": 0.1923542320728302,
1752
+ "learning_rate": 0.0009082129222406087,
1753
+ "loss": 2.843,
1754
+ "step": 22400
1755
+ },
1756
+ {
1757
+ "epoch": 0.2922077922077922,
1758
+ "grad_norm": 0.19237105548381805,
1759
+ "learning_rate": 0.0009073786822151326,
1760
+ "loss": 2.8554,
1761
+ "step": 22500
1762
+ },
1763
+ {
1764
+ "epoch": 0.2935064935064935,
1765
+ "grad_norm": 0.1831800788640976,
1766
+ "learning_rate": 0.000906541054970676,
1767
+ "loss": 2.8928,
1768
+ "step": 22600
1769
+ },
1770
+ {
1771
+ "epoch": 0.2948051948051948,
1772
+ "grad_norm": 0.20052975416183472,
1773
+ "learning_rate": 0.000905700047471832,
1774
+ "loss": 2.8462,
1775
+ "step": 22700
1776
+ },
1777
+ {
1778
+ "epoch": 0.2961038961038961,
1779
+ "grad_norm": 0.20518037676811218,
1780
+ "learning_rate": 0.0009048556667113002,
1781
+ "loss": 2.8624,
1782
+ "step": 22800
1783
+ },
1784
+ {
1785
+ "epoch": 0.2974025974025974,
1786
+ "grad_norm": 0.18524277210235596,
1787
+ "learning_rate": 0.0009040079197098268,
1788
+ "loss": 2.8683,
1789
+ "step": 22900
1790
+ },
1791
+ {
1792
+ "epoch": 0.2987012987012987,
1793
+ "grad_norm": 0.18116071820259094,
1794
+ "learning_rate": 0.000903156813516148,
1795
+ "loss": 2.8335,
1796
+ "step": 23000
1797
+ },
1798
+ {
1799
+ "epoch": 0.2987012987012987,
1800
+ "eval_loss": 3.236686944961548,
1801
+ "eval_runtime": 16.906,
1802
+ "eval_samples_per_second": 34.071,
1803
+ "eval_steps_per_second": 8.518,
1804
+ "step": 23000
1805
+ },
1806
+ {
1807
+ "epoch": 0.3,
1808
+ "grad_norm": 0.2058299481868744,
1809
+ "learning_rate": 0.0009023023552069303,
1810
+ "loss": 2.8548,
1811
+ "step": 23100
1812
+ },
1813
+ {
1814
+ "epoch": 0.3012987012987013,
1815
+ "grad_norm": 0.21021772921085358,
1816
+ "learning_rate": 0.0009014445518867116,
1817
+ "loss": 2.8435,
1818
+ "step": 23200
1819
+ },
1820
+ {
1821
+ "epoch": 0.3025974025974026,
1822
+ "grad_norm": 0.18895122408866882,
1823
+ "learning_rate": 0.000900583410687843,
1824
+ "loss": 2.839,
1825
+ "step": 23300
1826
+ },
1827
+ {
1828
+ "epoch": 0.3038961038961039,
1829
+ "grad_norm": 0.18983018398284912,
1830
+ "learning_rate": 0.0008997189387704286,
1831
+ "loss": 2.843,
1832
+ "step": 23400
1833
+ },
1834
+ {
1835
+ "epoch": 0.3051948051948052,
1836
+ "grad_norm": 0.21453820168972015,
1837
+ "learning_rate": 0.0008988511433222664,
1838
+ "loss": 2.8443,
1839
+ "step": 23500
1840
+ },
1841
+ {
1842
+ "epoch": 0.3064935064935065,
1843
+ "grad_norm": 0.1873568445444107,
1844
+ "learning_rate": 0.0008979800315587885,
1845
+ "loss": 2.8473,
1846
+ "step": 23600
1847
+ },
1848
+ {
1849
+ "epoch": 0.3077922077922078,
1850
+ "grad_norm": 0.19516156613826752,
1851
+ "learning_rate": 0.000897105610723001,
1852
+ "loss": 2.8579,
1853
+ "step": 23700
1854
+ },
1855
+ {
1856
+ "epoch": 0.3090909090909091,
1857
+ "grad_norm": 0.1961786150932312,
1858
+ "learning_rate": 0.000896227888085424,
1859
+ "loss": 2.8282,
1860
+ "step": 23800
1861
+ },
1862
+ {
1863
+ "epoch": 0.3103896103896104,
1864
+ "grad_norm": 0.19147805869579315,
1865
+ "learning_rate": 0.0008953468709440309,
1866
+ "loss": 2.8734,
1867
+ "step": 23900
1868
+ },
1869
+ {
1870
+ "epoch": 0.3116883116883117,
1871
+ "grad_norm": 0.20040926337242126,
1872
+ "learning_rate": 0.0008944625666241877,
1873
+ "loss": 2.8491,
1874
+ "step": 24000
1875
+ },
1876
+ {
1877
+ "epoch": 0.3116883116883117,
1878
+ "eval_loss": 3.2374658584594727,
1879
+ "eval_runtime": 16.1905,
1880
+ "eval_samples_per_second": 35.576,
1881
+ "eval_steps_per_second": 8.894,
1882
+ "step": 24000
1883
  }
1884
  ],
1885
  "logging_steps": 100,
 
1899
  "attributes": {}
1900
  }
1901
  },
1902
+ "total_flos": 7.1446353149952e+17,
1903
  "train_batch_size": 22,
1904
  "trial_name": null,
1905
  "trial_params": null