CodeIsAbstract commited on
Commit
f2ce635
·
verified ·
1 Parent(s): ba5c401

Training in progress, step 2500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bce9cd2ecb7520c5fbf1827487189abe146f8a04937f02a2e35fa41e6c94ae73
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d81ad66b4930d7b2dd698095062a3016fe78a7fa017e78cd010fb6336a5852a1
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a77095985c695d5aefdce15118c41ee7d9a582af6b20e0ad1dd0d0561b91b3f
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:45c86d2326cf881f6bfca4782a087b281584d3ed7d1eb8ee04cb9929d154a36b
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3136c830d048cb66eafa21e920652967c1c1c1766cfb212fe4e06f8320840cf4
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:325c636e6311e0fe4ca41b46abaf69ab835c7fa1be4ee8051f00fed0ade88598
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4a6e444c46ec49de792e4afbe9af4aa4613bca60425da2b0ac2cae225e516fcc
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b652e5527d5065c8d23b740b7553715cb3c1ddbdfe1b6360339df9f41f3fe84c
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 100,
7
- "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1568,6 +1568,396 @@
1568
  "eval_samples_per_second": 67.014,
1569
  "eval_steps_per_second": 3.823,
1570
  "step": 2000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1571
  }
1572
  ],
1573
  "logging_steps": 10,
@@ -1587,7 +1977,7 @@
1587
  "attributes": {}
1588
  }
1589
  },
1590
- "total_flos": 6.52309029715968e+17,
1591
  "train_batch_size": 18,
1592
  "trial_name": null,
1593
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.5,
6
  "eval_steps": 100,
7
+ "global_step": 2500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1568
  "eval_samples_per_second": 67.014,
1569
  "eval_steps_per_second": 3.823,
1570
  "step": 2000
1571
+ },
1572
+ {
1573
+ "epoch": 0.402,
1574
+ "grad_norm": 0.059326171875,
1575
+ "learning_rate": 0.0003,
1576
+ "loss": 2.608153533935547,
1577
+ "step": 2010
1578
+ },
1579
+ {
1580
+ "epoch": 0.404,
1581
+ "grad_norm": 0.20703125,
1582
+ "learning_rate": 0.0003,
1583
+ "loss": 2.6085268020629884,
1584
+ "step": 2020
1585
+ },
1586
+ {
1587
+ "epoch": 0.406,
1588
+ "grad_norm": 0.9296875,
1589
+ "learning_rate": 0.0003,
1590
+ "loss": 2.605429267883301,
1591
+ "step": 2030
1592
+ },
1593
+ {
1594
+ "epoch": 0.408,
1595
+ "grad_norm": 0.08056640625,
1596
+ "learning_rate": 0.0003,
1597
+ "loss": 2.6161428451538087,
1598
+ "step": 2040
1599
+ },
1600
+ {
1601
+ "epoch": 0.41,
1602
+ "grad_norm": 0.0576171875,
1603
+ "learning_rate": 0.0003,
1604
+ "loss": 2.587337112426758,
1605
+ "step": 2050
1606
+ },
1607
+ {
1608
+ "epoch": 0.412,
1609
+ "grad_norm": 1.5546875,
1610
+ "learning_rate": 0.0003,
1611
+ "loss": 2.613199996948242,
1612
+ "step": 2060
1613
+ },
1614
+ {
1615
+ "epoch": 0.414,
1616
+ "grad_norm": 0.4375,
1617
+ "learning_rate": 0.0003,
1618
+ "loss": 2.595570945739746,
1619
+ "step": 2070
1620
+ },
1621
+ {
1622
+ "epoch": 0.416,
1623
+ "grad_norm": 0.059326171875,
1624
+ "learning_rate": 0.0003,
1625
+ "loss": 2.614238166809082,
1626
+ "step": 2080
1627
+ },
1628
+ {
1629
+ "epoch": 0.418,
1630
+ "grad_norm": 0.050537109375,
1631
+ "learning_rate": 0.0003,
1632
+ "loss": 2.5990161895751953,
1633
+ "step": 2090
1634
+ },
1635
+ {
1636
+ "epoch": 0.42,
1637
+ "grad_norm": 0.05126953125,
1638
+ "learning_rate": 0.0003,
1639
+ "loss": 2.6036991119384765,
1640
+ "step": 2100
1641
+ },
1642
+ {
1643
+ "epoch": 0.42,
1644
+ "eval_loss": 3.0276641845703125,
1645
+ "eval_runtime": 4.4443,
1646
+ "eval_samples_per_second": 67.052,
1647
+ "eval_steps_per_second": 3.825,
1648
+ "step": 2100
1649
+ },
1650
+ {
1651
+ "epoch": 0.422,
1652
+ "grad_norm": 0.05126953125,
1653
+ "learning_rate": 0.0003,
1654
+ "loss": 2.5789426803588866,
1655
+ "step": 2110
1656
+ },
1657
+ {
1658
+ "epoch": 0.424,
1659
+ "grad_norm": 0.083984375,
1660
+ "learning_rate": 0.0003,
1661
+ "loss": 2.592574119567871,
1662
+ "step": 2120
1663
+ },
1664
+ {
1665
+ "epoch": 0.426,
1666
+ "grad_norm": 0.049072265625,
1667
+ "learning_rate": 0.0003,
1668
+ "loss": 2.592014122009277,
1669
+ "step": 2130
1670
+ },
1671
+ {
1672
+ "epoch": 0.428,
1673
+ "grad_norm": 0.0859375,
1674
+ "learning_rate": 0.0003,
1675
+ "loss": 2.5774700164794924,
1676
+ "step": 2140
1677
+ },
1678
+ {
1679
+ "epoch": 0.43,
1680
+ "grad_norm": 0.05615234375,
1681
+ "learning_rate": 0.0003,
1682
+ "loss": 2.613265037536621,
1683
+ "step": 2150
1684
+ },
1685
+ {
1686
+ "epoch": 0.432,
1687
+ "grad_norm": 0.068359375,
1688
+ "learning_rate": 0.0003,
1689
+ "loss": 2.5970441818237306,
1690
+ "step": 2160
1691
+ },
1692
+ {
1693
+ "epoch": 0.434,
1694
+ "grad_norm": 0.08447265625,
1695
+ "learning_rate": 0.0003,
1696
+ "loss": 2.6015340805053713,
1697
+ "step": 2170
1698
+ },
1699
+ {
1700
+ "epoch": 0.436,
1701
+ "grad_norm": 0.640625,
1702
+ "learning_rate": 0.0003,
1703
+ "loss": 2.58035945892334,
1704
+ "step": 2180
1705
+ },
1706
+ {
1707
+ "epoch": 0.438,
1708
+ "grad_norm": 0.06640625,
1709
+ "learning_rate": 0.0003,
1710
+ "loss": 2.609104537963867,
1711
+ "step": 2190
1712
+ },
1713
+ {
1714
+ "epoch": 0.44,
1715
+ "grad_norm": 0.057861328125,
1716
+ "learning_rate": 0.0003,
1717
+ "loss": 2.5889204025268553,
1718
+ "step": 2200
1719
+ },
1720
+ {
1721
+ "epoch": 0.44,
1722
+ "eval_loss": 3.0296566486358643,
1723
+ "eval_runtime": 4.4076,
1724
+ "eval_samples_per_second": 67.61,
1725
+ "eval_steps_per_second": 3.857,
1726
+ "step": 2200
1727
+ },
1728
+ {
1729
+ "epoch": 0.442,
1730
+ "grad_norm": 0.054931640625,
1731
+ "learning_rate": 0.0003,
1732
+ "loss": 2.582586097717285,
1733
+ "step": 2210
1734
+ },
1735
+ {
1736
+ "epoch": 0.444,
1737
+ "grad_norm": 0.06787109375,
1738
+ "learning_rate": 0.0003,
1739
+ "loss": 2.606948471069336,
1740
+ "step": 2220
1741
+ },
1742
+ {
1743
+ "epoch": 0.446,
1744
+ "grad_norm": 0.7421875,
1745
+ "learning_rate": 0.0003,
1746
+ "loss": 2.5844865798950196,
1747
+ "step": 2230
1748
+ },
1749
+ {
1750
+ "epoch": 0.448,
1751
+ "grad_norm": 0.1025390625,
1752
+ "learning_rate": 0.0003,
1753
+ "loss": 2.5672168731689453,
1754
+ "step": 2240
1755
+ },
1756
+ {
1757
+ "epoch": 0.45,
1758
+ "grad_norm": 0.05810546875,
1759
+ "learning_rate": 0.0003,
1760
+ "loss": 2.588663673400879,
1761
+ "step": 2250
1762
+ },
1763
+ {
1764
+ "epoch": 0.452,
1765
+ "grad_norm": 0.05224609375,
1766
+ "learning_rate": 0.0003,
1767
+ "loss": 2.594828987121582,
1768
+ "step": 2260
1769
+ },
1770
+ {
1771
+ "epoch": 0.454,
1772
+ "grad_norm": 0.060546875,
1773
+ "learning_rate": 0.0003,
1774
+ "loss": 2.6015480041503904,
1775
+ "step": 2270
1776
+ },
1777
+ {
1778
+ "epoch": 0.456,
1779
+ "grad_norm": 0.0576171875,
1780
+ "learning_rate": 0.0003,
1781
+ "loss": 2.6068796157836913,
1782
+ "step": 2280
1783
+ },
1784
+ {
1785
+ "epoch": 0.458,
1786
+ "grad_norm": 0.5859375,
1787
+ "learning_rate": 0.0003,
1788
+ "loss": 2.59091854095459,
1789
+ "step": 2290
1790
+ },
1791
+ {
1792
+ "epoch": 0.46,
1793
+ "grad_norm": 0.1142578125,
1794
+ "learning_rate": 0.0003,
1795
+ "loss": 2.5947988510131834,
1796
+ "step": 2300
1797
+ },
1798
+ {
1799
+ "epoch": 0.46,
1800
+ "eval_loss": 3.028251886367798,
1801
+ "eval_runtime": 4.4313,
1802
+ "eval_samples_per_second": 67.249,
1803
+ "eval_steps_per_second": 3.836,
1804
+ "step": 2300
1805
+ },
1806
+ {
1807
+ "epoch": 0.462,
1808
+ "grad_norm": 0.0673828125,
1809
+ "learning_rate": 0.0003,
1810
+ "loss": 2.604340744018555,
1811
+ "step": 2310
1812
+ },
1813
+ {
1814
+ "epoch": 0.464,
1815
+ "grad_norm": 0.08203125,
1816
+ "learning_rate": 0.0003,
1817
+ "loss": 2.610575866699219,
1818
+ "step": 2320
1819
+ },
1820
+ {
1821
+ "epoch": 0.466,
1822
+ "grad_norm": 0.26953125,
1823
+ "learning_rate": 0.0003,
1824
+ "loss": 2.5987600326538085,
1825
+ "step": 2330
1826
+ },
1827
+ {
1828
+ "epoch": 0.468,
1829
+ "grad_norm": 0.1572265625,
1830
+ "learning_rate": 0.0003,
1831
+ "loss": 2.5745519638061523,
1832
+ "step": 2340
1833
+ },
1834
+ {
1835
+ "epoch": 0.47,
1836
+ "grad_norm": 0.0751953125,
1837
+ "learning_rate": 0.0003,
1838
+ "loss": 2.5737417221069334,
1839
+ "step": 2350
1840
+ },
1841
+ {
1842
+ "epoch": 0.472,
1843
+ "grad_norm": 0.061279296875,
1844
+ "learning_rate": 0.0003,
1845
+ "loss": 2.586330795288086,
1846
+ "step": 2360
1847
+ },
1848
+ {
1849
+ "epoch": 0.474,
1850
+ "grad_norm": 0.04931640625,
1851
+ "learning_rate": 0.0003,
1852
+ "loss": 2.6139583587646484,
1853
+ "step": 2370
1854
+ },
1855
+ {
1856
+ "epoch": 0.476,
1857
+ "grad_norm": 0.05908203125,
1858
+ "learning_rate": 0.0003,
1859
+ "loss": 2.55531005859375,
1860
+ "step": 2380
1861
+ },
1862
+ {
1863
+ "epoch": 0.478,
1864
+ "grad_norm": 0.060302734375,
1865
+ "learning_rate": 0.0003,
1866
+ "loss": 2.601852226257324,
1867
+ "step": 2390
1868
+ },
1869
+ {
1870
+ "epoch": 0.48,
1871
+ "grad_norm": 0.05712890625,
1872
+ "learning_rate": 0.0003,
1873
+ "loss": 2.6031288146972655,
1874
+ "step": 2400
1875
+ },
1876
+ {
1877
+ "epoch": 0.48,
1878
+ "eval_loss": 3.02764892578125,
1879
+ "eval_runtime": 4.4578,
1880
+ "eval_samples_per_second": 66.849,
1881
+ "eval_steps_per_second": 3.814,
1882
+ "step": 2400
1883
+ },
1884
+ {
1885
+ "epoch": 0.482,
1886
+ "grad_norm": 0.053466796875,
1887
+ "learning_rate": 0.0003,
1888
+ "loss": 2.5988203048706056,
1889
+ "step": 2410
1890
+ },
1891
+ {
1892
+ "epoch": 0.484,
1893
+ "grad_norm": 0.05517578125,
1894
+ "learning_rate": 0.0003,
1895
+ "loss": 2.5995391845703124,
1896
+ "step": 2420
1897
+ },
1898
+ {
1899
+ "epoch": 0.486,
1900
+ "grad_norm": 0.054443359375,
1901
+ "learning_rate": 0.0003,
1902
+ "loss": 2.5926639556884767,
1903
+ "step": 2430
1904
+ },
1905
+ {
1906
+ "epoch": 0.488,
1907
+ "grad_norm": 0.050537109375,
1908
+ "learning_rate": 0.0003,
1909
+ "loss": 2.615771484375,
1910
+ "step": 2440
1911
+ },
1912
+ {
1913
+ "epoch": 0.49,
1914
+ "grad_norm": 0.216796875,
1915
+ "learning_rate": 0.0003,
1916
+ "loss": 2.5967260360717774,
1917
+ "step": 2450
1918
+ },
1919
+ {
1920
+ "epoch": 0.492,
1921
+ "grad_norm": 0.1171875,
1922
+ "learning_rate": 0.0003,
1923
+ "loss": 2.6070770263671874,
1924
+ "step": 2460
1925
+ },
1926
+ {
1927
+ "epoch": 0.494,
1928
+ "grad_norm": 0.04931640625,
1929
+ "learning_rate": 0.0003,
1930
+ "loss": 2.632620620727539,
1931
+ "step": 2470
1932
+ },
1933
+ {
1934
+ "epoch": 0.496,
1935
+ "grad_norm": 0.08056640625,
1936
+ "learning_rate": 0.0003,
1937
+ "loss": 2.6182647705078126,
1938
+ "step": 2480
1939
+ },
1940
+ {
1941
+ "epoch": 0.498,
1942
+ "grad_norm": 0.0537109375,
1943
+ "learning_rate": 0.0003,
1944
+ "loss": 2.6368099212646485,
1945
+ "step": 2490
1946
+ },
1947
+ {
1948
+ "epoch": 0.5,
1949
+ "grad_norm": 0.80078125,
1950
+ "learning_rate": 0.0003,
1951
+ "loss": 2.607630729675293,
1952
+ "step": 2500
1953
+ },
1954
+ {
1955
+ "epoch": 0.5,
1956
+ "eval_loss": 3.0262646675109863,
1957
+ "eval_runtime": 4.4503,
1958
+ "eval_samples_per_second": 66.962,
1959
+ "eval_steps_per_second": 3.82,
1960
+ "step": 2500
1961
  }
1962
  ],
1963
  "logging_steps": 10,
 
1977
  "attributes": {}
1978
  }
1979
  },
1980
+ "total_flos": 8.1538628714496e+17,
1981
  "train_batch_size": 18,
1982
  "trial_name": null,
1983
  "trial_params": null